From f4561bd153af1d61115d09ec9e76d01d70ae6aa9 Mon Sep 17 00:00:00 2001 From: Victor Garcia Date: Mon, 28 Sep 2026 00:30:54 -0600 Subject: [PATCH 1/8] feat(worker): re-run flaky Actions checks before spending a repair round (U4) publish.ci.rerun: {budget: N} (1-3, requires wait) re-runs failed GitHub Actions jobs on the head jig pushed when CI is red, before any repair round or the attempt's end. The step waits until nothing on the head is pending, freshens the lease before each `gh api -X POST .../actions/jobs/{id}/rerun`, waits out "in progress" refusals without spending budget, and re-judges CI reading each re-run check run as pending until a new run replaces it. Every re-run lands in the summary's ci_reruns [{attempt, jobs, outcome}]; a pass after one sets ci_flaky. Re-runs also apply inside repairCI's loop, within the per-attempt budget. factory.yaml opts in with budget 1. Co-Authored-By: Claude Opus 5.5 --- README.md | 21 + cmd/jig/ci_repair_test.go | 6 + cmd/jig/factory_test.go | 4 + examples/definitions/factory.yaml | 1 + internal/protocol/definition.go | 36 +- internal/protocol/definition_test.go | 30 ++ internal/worker/publish.go | 114 ++++- internal/worker/publish_repair.go | 3 + internal/worker/publish_rerun.go | 335 ++++++++++++++ internal/worker/publish_rerun_test.go | 637 ++++++++++++++++++++++++++ internal/worker/publish_test.go | 21 + 11 files changed, 1191 insertions(+), 17 deletions(-) create mode 100644 internal/worker/publish_rerun.go create mode 100644 internal/worker/publish_rerun_test.go diff --git a/README.md b/README.md index 9c47fb5..e65ff6e 100644 --- a/README.md +++ b/README.md @@ -256,6 +256,27 @@ Rules worth knowing before you write one: code is the risk to design for: `factory.yaml` scores edits to CI or lint configuration, and test files that lose more lines than they gain, as not low, so such a "fix" is held for a person. +- **A flaky Actions job can be re-run before a round is spent.** + `publish: {ci: {wait: true, rerun: {budget: 1}}}` re-runs the failed + GitHub Actions jobs on the same head when CI is red on a head jig pushed, + before any repair round (or before the job ends red, with no `on_fail`). + It first waits, within the CI timeout, until nothing on the head is still + running, since GitHub will not re-run a job in a running workflow; then it + re-runs each failed job with `gh api -X POST + repos/{owner}/{repo}/actions/jobs/{id}/rerun`, freshening the lease before + every request, and awaits CI again, reading each re-run job as pending + until its new run replaces the old red one. GitHub refusing because a + workflow run is in progress is waited out and does not spend the budget; + any other refusal is recorded and falls through to repair or the end. If + any red check is not an Actions job (a commit status, another app), no + re-run happens. Re-runs never move the branch. The budget, 1–3, is per + attempt: a repair round's new head gets only what is left. Every re-run is + in the result's `ci_reruns` (`attempt`, `head`, `jobs`, `outcome`), and a + pass after one sets `ci_flaky` — it is reported as flaky, never as a clean + pass. Re-run waits run on wall-clock time, so time spent on them is time a + later repair round no longer has under the attempt's ceiling. The + publish-only retry never re-runs. `gh` needs permission to + re-run Actions jobs on the repository. - **Validation happens before anything runs.** `jig def validate ` is the same check the store applies at save time, offline. diff --git a/cmd/jig/ci_repair_test.go b/cmd/jig/ci_repair_test.go index 0f6256a..4216dbb 100644 --- a/cmd/jig/ci_repair_test.go +++ b/cmd/jig/ci_repair_test.go @@ -75,6 +75,12 @@ func (g *e2eGateway) FailedCheckLogs(_ context.Context, _ string, checks []worke return checks } +// RerunActionsJob is never reached: the definition declares no re-runs, and +// its checks are not Actions jobs. +func (g *e2eGateway) RerunActionsJob(context.Context, string, int64) error { + return fmt.Errorf("unexpected re-run request") +} + func TestACIRepairRoundRunsThroughTheRealWorkerWiring(t *testing.T) { repo := initRepo(t) serverData, workerData := t.TempDir(), t.TempDir() diff --git a/cmd/jig/factory_test.go b/cmd/jig/factory_test.go index be0ddc6..464ab1a 100644 --- a/cmd/jig/factory_test.go +++ b/cmd/jig/factory_test.go @@ -141,6 +141,10 @@ func TestTheFactoryRepairsRedCIThroughItsWholeReviewPanel(t *testing.T) { spec.Publish.CI.OnFail.Budget < 1 { t.Fatalf("publish.ci = %+v, want on_fail repairing through build", spec.Publish.CI) } + // One re-run tells a flaky job from a broken one before a round is spent. + if spec.Publish.CI.Rerun == nil || spec.Publish.CI.Rerun.Budget != 1 { + t.Fatalf("publish.ci.rerun = %+v, want budget 1", spec.Publish.CI.Rerun) + } after := map[string]bool{} seen := false for _, phase := range spec.Phases { diff --git a/examples/definitions/factory.yaml b/examples/definitions/factory.yaml index 25e562f..2363a7b 100644 --- a/examples/definitions/factory.yaml +++ b/examples/definitions/factory.yaml @@ -387,5 +387,6 @@ publish: hold_when: "risk != low" ci: wait: true + rerun: {budget: 1} on_fail: {run: build, budget: 2} timeout: 30m diff --git a/internal/protocol/definition.go b/internal/protocol/definition.go index d749a06..7701464 100644 --- a/internal/protocol/definition.go +++ b/internal/protocol/definition.go @@ -83,11 +83,24 @@ type PublishSpec struct { // CISpec opts a definition into waiting for the pull request's CI after // publish. Timeout is a Go duration ("30m"); empty means DefaultCITimeout. // OnFail, when declared, repairs red CI inside the attempt instead of ending -// it there. +// it there. Rerun, when declared, re-runs failed GitHub Actions jobs on the +// same head before any repair round, to tell a flaky check from a real one. type CISpec struct { Wait bool `yaml:"wait"` Timeout string `yaml:"timeout"` OnFail *CIRepairSpec `yaml:"on_fail"` + Rerun *CIRerunSpec `yaml:"rerun"` +} + +// CIRerunSpec is the flaky-check re-run policy: +// +// rerun: {budget: N} +// +// Budget counts re-runs per attempt, not per head: a repair round's new head +// gets only what the earlier heads left. A re-run never changes the branch, +// and a pass after one is reported as flaky, never as a clean pass. +type CIRerunSpec struct { + Budget int `yaml:"budget"` } // CIRepairSpec is the CI repair loop: @@ -115,6 +128,10 @@ const ( // every phase after the repair phase, reviewers included, so a // larger budget is mostly a larger bill for a fix that is not converging. MaxCIRepairRounds = 3 + + // MaxCIReruns caps publish.ci.rerun's budget. Re-running more often than + // this stops telling a flaky check from a broken one and starts hiding it. + MaxCIReruns = 3 ) // WaitsForCI reports whether the definition gates acceptance on CI. @@ -315,9 +332,26 @@ func (spec *DefinitionSpec) validatePublish(phases map[string]PhaseSpec) error { timeout, MinCITimeout, MaxCITimeout) } } + if err := spec.validateCIRerun(); err != nil { + return err + } return spec.validateCIRepair(phases) } +func (spec *DefinitionSpec) validateCIRerun() error { + ci := spec.Publish.CI + if ci == nil || ci.Rerun == nil { + return nil + } + if !ci.Wait { + return fmt.Errorf("publish: ci: rerun re-runs red CI checks, which needs wait: true") + } + if ci.Rerun.Budget < 1 || ci.Rerun.Budget > MaxCIReruns { + return fmt.Errorf("publish: ci: rerun: budget %d is outside 1..%d", ci.Rerun.Budget, MaxCIReruns) + } + return nil +} + func (spec *DefinitionSpec) validateCIRepair(phases map[string]PhaseSpec) error { ci := spec.Publish.CI if ci == nil || ci.OnFail == nil { diff --git a/internal/protocol/definition_test.go b/internal/protocol/definition_test.go index 9b912ab..9811b4d 100644 --- a/internal/protocol/definition_test.go +++ b/internal/protocol/definition_test.go @@ -525,6 +525,36 @@ phases: "resume_from") } +func TestPublishCIRerunIsBoundedAndNeedsAWait(t *testing.T) { + const base = ` +name: ci-rerun +roster: + builder: {model: opus, system_prompt: s, user_prompt: u} +phases: + - {name: build, kind: agent, owner: builder} +` + if spec := mustParse(t, base+"publish: {ci: {wait: true}}\n"); spec.Publish.CI.Rerun != nil { + t.Fatalf("rerun = %+v, want nil when undeclared", spec.Publish.CI.Rerun) + } + for budget := 1; budget <= MaxCIReruns; budget++ { + spec := mustParse(t, base+fmt.Sprintf("publish: {ci: {wait: true, rerun: {budget: %d}}}\n", budget)) + if spec.Publish.CI.Rerun == nil || spec.Publish.CI.Rerun.Budget != budget { + t.Fatalf("rerun = %+v, want budget %d", spec.Publish.CI.Rerun, budget) + } + } + if MaxCIReruns != 3 { + t.Fatalf("MaxCIReruns = %d, want 3", MaxCIReruns) + } + mustReject(t, base+"publish: {ci: {wait: false, rerun: {budget: 1}}}\n", + "publish: ci: rerun", "wait: true") + mustReject(t, base+"publish: {ci: {rerun: {budget: 1}}}\n", + "publish: ci: rerun", "wait: true") + mustReject(t, base+"publish: {ci: {wait: true, rerun: {}}}\n", + "publish: ci: rerun: budget 0", "outside 1..3") + mustReject(t, base+"publish: {ci: {wait: true, rerun: {budget: 4}}}\n", + "publish: ci: rerun: budget 4", "outside 1..3") +} + func TestNegativeRoleBudgetIsRejected(t *testing.T) { spec := mustParse(t, ` name: budget diff --git a/internal/worker/publish.go b/internal/worker/publish.go index 02cfdcf..1e6ca58 100644 --- a/internal/worker/publish.go +++ b/internal/worker/publish.go @@ -127,8 +127,12 @@ type PublishSummary struct { CIFailures []CICheck `json:"ci_failures,omitempty"` // CIRepairs lists the CI repair rounds this publish ran, in order. CIRepairs []CIRepairSummary `json:"ci_repairs,omitempty"` - Performed []string `json:"performed_steps,omitempty"` - Reused []string `json:"reused_steps,omitempty"` + // CIReruns lists the flaky-check re-runs this publish ran, in order; + // CIFlaky says CI went green only after one (R6), never a clean pass. + CIReruns []CIRerunSummary `json:"ci_reruns,omitempty"` + CIFlaky bool `json:"ci_flaky,omitempty"` + Performed []string `json:"performed_steps,omitempty"` + Reused []string `json:"reused_steps,omitempty"` } // Published reports whether the pipeline completed with remote proof — the @@ -171,6 +175,11 @@ type PullRequestGateway interface { // job's log attached, within the CI repair log bounds. It never fails: // a log that cannot be read leaves a LogNote saying why. FailedCheckLogs(ctx context.Context, repository string, checks []CICheck) []CICheck + // RerunActionsJob asks GitHub to re-run one failed Actions job on the + // same commit. A refusal because the job's workflow run is still in + // progress carries the code ci_rerun_in_progress, so it can be waited + // out rather than counted. + RerunActionsJob(ctx context.Context, repository string, jobID int64) error } // CI check verdicts, normalized across check runs and commit statuses. @@ -240,6 +249,9 @@ type ciPolicy struct { // repairBudget is publish.ci.on_fail's budget; zero means red CI ends // the attempt, as it always has. repairBudget int + // rerunBudget is publish.ci.rerun's budget: flaky-check re-runs per + // attempt; zero means none. + rerunBudget int } // ciPolicyFor reads the policy out of a run snapshot. A snapshot that does @@ -255,6 +267,9 @@ func ciPolicyFor(snapshot string) ciPolicy { if policy.wait && spec.Publish.CI.OnFail != nil { policy.repairBudget = spec.Publish.CI.OnFail.Budget } + if policy.wait && spec.Publish.CI.Rerun != nil { + policy.rerunBudget = spec.Publish.CI.Rerun.Budget + } return policy } @@ -361,6 +376,9 @@ func (r *PublishingRunner) Run(ctx context.Context, prepared *PreparedAttempt) O ci := ciPolicyFor(prepared.Claim.Snapshot) summary := worker.publish(context.WithoutCancel(ctx), r.gateway, r.options, target, changedPathsFromResult(outcome.Result), ci) + // A flaky Actions job gets its declared re-runs before red CI costs a + // repair round or ends the attempt (R5). Re-runs never move the branch. + summary = worker.rerunFlakyCI(context.WithoutCancel(ctx), r.gateway, r.options, target, summary, ci) if summary.Code == "ci_failed" && outcome.Continuation != nil && ci.repairBudget > 0 { // Rounds run on the same uncancelled context: cancellation reaches // the engine through the attempt's cancel channel and the CI wait @@ -731,6 +749,16 @@ func (w *Worker) remoteHead(ctx context.Context, target publishTarget, branch st func (w *Worker) awaitCI( ctx context.Context, gateway PullRequestGateway, options PublishOptions, target publishTarget, branch string, timeout time.Duration, +) (string, []CICheck, error) { + return w.awaitCIViewed(ctx, gateway, options, target, branch, timeout, nil) +} + +// awaitCIViewed is awaitCI reading each poll through a re-run view, so a +// check run jig just re-ran counts as pending until its new run replaces it +// rather than being read back red (KTD5). A nil view is the plain wait. +func (w *Worker) awaitCIViewed( + ctx context.Context, gateway PullRequestGateway, options PublishOptions, + target publishTarget, branch string, timeout time.Duration, view *rerunView, ) (string, []CICheck, error) { start := time.Now() deadline := start.Add(timeout) @@ -761,6 +789,7 @@ func (w *Worker) awaitCI( } } else { transient = 0 + checks = view.apply(checks) var failed []CICheck pending = pending[:0] for _, check := range checks { @@ -790,24 +819,34 @@ func (w *Worker) awaitCI( return head, boundedChecks(pending), publishFailure("ci_timeout", "CI did not finish on %s within %s; %s", shortSHA(head), timeout, waiting) } - wait := options.ciPollInterval() - if remaining := time.Until(deadline); remaining < wait { - wait = remaining - } - timer := time.NewTimer(wait) - select { - case <-target.lease.cancelled: - timer.Stop() - return head, nil, publishFailure("ci_wait_cancelled", - "the job was cancelled while waiting for CI on %s", shortSHA(head)) - case <-ctx.Done(): - timer.Stop() - return head, nil, publishFailure("ci_wait_cancelled", "the CI wait was interrupted: %s", ctx.Err()) - case <-timer.C: + if err := w.ciPause(ctx, options, target, head, deadline); err != nil { + return head, nil, err } } } +// ciPause sleeps one poll interval, or until the deadline if that is +// sooner, and ends early with ci_wait_cancelled when the job is cancelled. +func (w *Worker) ciPause( + ctx context.Context, options PublishOptions, target publishTarget, head string, deadline time.Time, +) error { + wait := options.ciPollInterval() + if remaining := time.Until(deadline); remaining < wait { + wait = remaining + } + timer := time.NewTimer(wait) + defer timer.Stop() + select { + case <-target.lease.cancelled: + return publishFailure("ci_wait_cancelled", + "the job was cancelled while waiting for CI on %s", shortSHA(head)) + case <-ctx.Done(): + return publishFailure("ci_wait_cancelled", "the CI wait was interrupted: %s", ctx.Err()) + case <-timer.C: + return nil + } +} + // maxReportedCIChecks bounds how many checks a summary names. const maxReportedCIChecks = 20 @@ -1523,6 +1562,49 @@ func (g *GitHubCLIGateway) FailedCheckLogs(ctx context.Context, repository strin return annotated } +// RerunActionsJob re-runs one failed Actions job with +// `gh api -X POST repos/{project}/actions/jobs/{id}/rerun`: per job, on the +// same commit, never moving the branch (KTD5). GitHub refuses while the +// job's workflow run is still going; that refusal is ci_rerun_in_progress. +func (g *GitHubCLIGateway) RerunActionsJob(ctx context.Context, repository string, jobID int64) error { + project, err := githubProject(repository) + if err != nil { + return err + } + if jobID <= 0 { + return publishFailure("ci_rerun_invalid_job", "%d is not an Actions job id", jobID) + } + if _, err := g.lookPath()("gh"); err != nil { + return publishFailure("gh_not_found", + "the GitHub CLI (gh) was not found on PATH. Install gh, then run `gh auth login`.") + } + stdout, stderr, stdoutTooLarge, stderrTooLarge, err := g.run()(ctx, "gh", + "api", "-X", "POST", "-H", "Accept: application/vnd.github+json", + fmt.Sprintf("repos/%s/actions/jobs/%d/rerun", project, jobID)) + if err == nil { + return nil + } + if rerunInProgress(stdout) || rerunInProgress(stderr) { + return publishFailure(ciRerunInProgress, + "GitHub will not re-run job %d while its workflow run is in progress: %s", jobID, + boundedText(strings.TrimSpace(string(stderr)), protocol.MaxPublishDiagnosticBytes)) + } + return ghDiagnostic("gh api job rerun", err, stderr, stdoutTooLarge, stderrTooLarge) +} + +// rerunInProgress recognizes GitHub's refusal to re-run a job whose +// workflow run has not completed. gh prints the API message on stderr and +// the response body on stdout; either may carry it. +func rerunInProgress(output []byte) bool { + lower := strings.ToLower(string(output)) + for _, phrase := range []string{"already running", "in progress", "is running", "not complete"} { + if strings.Contains(lower, phrase) { + return true + } + } + return false +} + // maxCILogReads bounds the log reads one FailedCheckLogs call attempts, // failures included: twice the kept logs leaves room for expired or // unreadable ones without letting a hanging gh cost more than diff --git a/internal/worker/publish_repair.go b/internal/worker/publish_repair.go index 79abae3..bb57a72 100644 --- a/internal/worker/publish_repair.go +++ b/internal/worker/publish_repair.go @@ -168,6 +168,9 @@ func (w *Worker) repairCI( summary.State, summary.Code, summary.Detail = PublishStatePublished, "", "" return summary } + // Red on the round's own head: whatever re-run budget the earlier + // heads left applies here before the next round is spent (R5). + summary = w.rerunFlakyCI(ctx, gateway, options, target, summary, ci) } return summary } diff --git a/internal/worker/publish_rerun.go b/internal/worker/publish_rerun.go new file mode 100644 index 0000000..639c586 --- /dev/null +++ b/internal/worker/publish_rerun.go @@ -0,0 +1,335 @@ +// publish_rerun.go — declared re-runs of flaky GitHub Actions checks +// (publish.ci.rerun, plan U4, KTD5). When CI is red on the head jig pushed +// and every red check is an Actions job, jig re-runs those jobs on the same +// head before it spends a repair round or ends the attempt: +// +// red CI → every red check an Actions job, budget left? +// → wait until nothing on the head is pending (GitHub refuses to +// re-run a job whose workflow run is still going) +// → freshen the lease, re-run each failed job (an "in progress" +// refusal waits and asks again; it spends nothing) +// → a fresh ci step on the same head, the re-run check runs pending +// until newer runs of the same name replace them +// → green publishes, flagged flaky; red loops while budget remains +// +// A re-run never moves the branch, so it is fenced by the lease rather than +// the ledger, and every one is recorded in the summary's ci_reruns. The +// budget is per attempt: a repair round's new head gets only what is left. +package worker + +import ( + "context" + "fmt" + "time" + + "github.com/StructuPath/jig/internal/protocol" +) + +// CIRerunSummary is one re-run as the publish summary reports it. The field +// names are pinned: the report reads ci_reruns[].{attempt, jobs, outcome}. +type CIRerunSummary struct { + // Attempt counts re-runs within the attempt, from 1. + Attempt int `json:"attempt"` + // Head is the commit the jobs were re-run on. + Head string `json:"head"` + // Jobs names the failed jobs re-run (or, when none were sent, the ones + // that would have been). + Jobs []string `json:"jobs"` + // Outcome is passed, failed, or the code the re-run stopped on. + Outcome string `json:"outcome"` + Detail string `json:"detail,omitempty"` +} + +const ( + ciRerunPassed = "passed" + ciRerunFailed = "failed" + // ciRerunRefused: GitHub refused a re-run for a reason other than a + // workflow run still in progress. + ciRerunRefused = "ci_rerun_refused" + // ciRerunHeadMoved: someone pushed to the branch while jig waited to + // re-run; there is nothing of jig's left to re-run. + ciRerunHeadMoved = "ci_rerun_head_moved" + // ciRerunInProgress is the gateway's code for GitHub refusing a re-run + // because the job's workflow run has not finished. It is waited out, + // never recorded as a spent re-run. + ciRerunInProgress = "ci_rerun_in_progress" + // maxCIRerunInProgressRetries bounds how often one job's re-run is asked + // for again after an "in progress" refusal; the CI timeout bounds how + // long. + maxCIRerunInProgressRetries = 5 +) + +// isActionsJob reports whether a check is a GitHub Actions job jig can +// re-run: only those have a job id, and it is their check-run id. +func isActionsJob(check CICheck) bool { + return check.App == "github-actions" && check.CheckRunID > 0 +} + +// allActionsJobs reports whether every red check is a re-runnable Actions +// job. One red check of any other kind means a re-run cannot turn CI green, +// so none is attempted (R7). +func allActionsJobs(checks []CICheck) bool { + if len(checks) == 0 { + return false + } + for _, check := range checks { + if check.Verdict == CIFail && !isActionsJob(check) { + return false + } + } + return true +} + +// rerunView is how the CI wait reads a head after re-runs: a check run jig +// just re-ran still shows its old red result until GitHub creates the new +// run, so it counts as pending until a check run of the same name that was +// not on the head before the re-run appears (KTD5). +type rerunView struct { + known map[int64]bool // every check-run id on the head before the re-run + rerun map[int64]bool // the check-run ids jig asked GitHub to re-run +} + +func newRerunView(checks []CICheck) *rerunView { + view := &rerunView{known: make(map[int64]bool, len(checks)), rerun: map[int64]bool{}} + for _, check := range checks { + if check.CheckRunID > 0 { + view.known[check.CheckRunID] = true + } + } + return view +} + +// apply rewrites one poll's checks. A re-run check is dropped once as many +// new runs of its name exist as were re-run under it, and is pending until +// then. A nil view changes nothing. +func (v *rerunView) apply(checks []CICheck) []CICheck { + if v == nil || len(v.rerun) == 0 { + return checks + } + fresh, stale := map[string]int{}, map[string]int{} + for _, check := range checks { + switch { + case v.rerun[check.CheckRunID]: + stale[check.Name]++ + case check.CheckRunID > 0 && !v.known[check.CheckRunID]: + fresh[check.Name]++ + } + } + viewed := make([]CICheck, 0, len(checks)) + for _, check := range checks { + if v.rerun[check.CheckRunID] { + if fresh[check.Name] >= stale[check.Name] { + continue + } + check.Verdict = CIPending + } + viewed = append(viewed, check) + } + return viewed +} + +// rerunFlakyCI runs declared re-runs while CI is red on the head jig pushed, +// every red check is an Actions job, and budget remains. summary is a +// publish summary that ended on ci_failed; the returned one is either +// published (flaky), still ci_failed (for a repair round or the end), or +// ended on the code a re-run stopped on. Every pass through the loop records +// one ci_reruns entry or returns, so it runs at most the budget's times. +func (w *Worker) rerunFlakyCI( + ctx context.Context, + gateway PullRequestGateway, + options PublishOptions, + target publishTarget, + summary PublishSummary, + ci ciPolicy, +) PublishSummary { + for summary.Code == "ci_failed" && len(summary.CIReruns) < ci.rerunBudget { + head := summary.CIRef + if head == "" || head != summary.RemoteRef || !allActionsJobs(summary.CIFailures) { + // Red on a head jig did not push is a person's to fix; red on a + // check that is not an Actions job cannot be re-run. + return summary + } + entry := CIRerunSummary{Attempt: len(summary.CIReruns) + 1, Head: head, + Jobs: checkNames(summary.CIFailures)} + deadline := time.Now().Add(ci.timeout) + + // The CI wait stopped at the first red check; its siblings may still + // be running, and GitHub will not re-run a job in a running workflow. + checks, err := w.settleCI(ctx, gateway, options, target, head, deadline, nil) + if err != nil { + return rerunStopped(summary, entry, err) + } + failed := failedChecks(checks) + if len(failed) == 0 { + // Nothing is red any more (someone re-ran it). Judge the head + // afresh; with no re-run of jig's, a pass here is not flaky. + return w.judgeAfterRerun(ctx, gateway, options, target, summary, ci, nil) + } + summary.CIFailures = boundedChecks(failed) + entry.Jobs = checkNames(failed) + if !allActionsJobs(failed) { + // A sibling that finished red is not an Actions job. + return summary + } + + view := newRerunView(checks) + if err := w.requestReruns(ctx, gateway, options, target, head, deadline, summary.CIFailures, view); err != nil { + return rerunStopped(summary, entry, err) + } + summary = w.judgeAfterRerun(ctx, gateway, options, target, summary, ci, view) + switch { + case summary.Published(): + summary.CIFlaky = true + entry.Outcome = ciRerunPassed + case summary.Code == "ci_failed": + entry.Outcome = ciRerunFailed + entry.Detail = summary.Detail + default: + entry.Outcome = summary.Code + entry.Detail = summary.Detail + } + summary.CIReruns = append(summary.CIReruns, entry) + } + return summary +} + +// rerunStopped records a re-run that stopped before its CI wait. A lost +// lease, a cancellation, and a moved head end the publish on that code; any +// other stop (a refusal, CI that never settled) leaves CI red, so a repair +// round or the attempt's end takes it from there. Either way the entry is +// recorded, and the loop does not go round again. +func rerunStopped(summary PublishSummary, entry CIRerunSummary, err error) PublishSummary { + code := publishCode(err) + entry.Outcome = code + entry.Detail = boundedText(err.Error(), protocol.MaxPublishDiagnosticBytes) + summary.CIReruns = append(summary.CIReruns, entry) + switch code { + case ciRerunRefused, "ci_timeout", "ci_unavailable": + return summary + } + return failSummary(summary, code, err.Error()) +} + +// requestReruns asks GitHub to re-run each failed job, freshening the lease +// before every request so an attempt that lost its lease sends none. An "in +// progress" refusal is waited out and asked again, within the deadline and +// maxCIRerunInProgressRetries; any other refusal stops the re-run. +func (w *Worker) requestReruns( + ctx context.Context, gateway PullRequestGateway, options PublishOptions, + target publishTarget, head string, deadline time.Time, failed []CICheck, view *rerunView, +) error { + for _, job := range failed { + for retries := 0; ; retries++ { + if err := target.lease.freshen(ctx); err != nil { + return fmt.Errorf("lease could not be freshened: %w", err) + } + err := gateway.RerunActionsJob(ctx, target.repository, job.CheckRunID) + if err == nil { + view.rerun[job.CheckRunID] = true + break + } + if publishCode(err) != ciRerunInProgress { + return publishFailure(ciRerunRefused, "GitHub refused to re-run %s: %s", job.Name, err.Error()) + } + if retries == maxCIRerunInProgressRetries { + return publishFailure(ciRerunRefused, + "GitHub still reported %s's workflow run in progress after %d waits: %s", + job.Name, retries, err.Error()) + } + if err := w.ciPause(ctx, options, target, head, deadline); err != nil { + return err + } + if _, err := w.settleCI(ctx, gateway, options, target, head, deadline, view); err != nil { + return err + } + } + } + return nil +} + +// judgeAfterRerun is a fresh, fenced ci step on the branch: green is +// recorded and publishes, anything else ends the step as the CI wait does. +func (w *Worker) judgeAfterRerun( + ctx context.Context, gateway PullRequestGateway, options PublishOptions, + target publishTarget, summary PublishSummary, ci ciPolicy, view *rerunView, +) PublishSummary { + summary.CIFailures = nil + green, err := w.publishStep(ctx, target, protocol.PublishStepCI, map[string]protocol.PublishRecord{}, &summary, + func(authorization protocol.PublishAuthorization) (protocol.PublishStepRequest, error) { + ciHead, failures, err := w.awaitCIViewed(ctx, gateway, options, target, authorization.Branch, + ci.timeout, view) + summary.CIRef = ciHead + summary.CIFailures = failures + if err != nil { + return protocol.PublishStepRequest{}, err + } + return protocol.PublishStepRequest{Step: protocol.PublishStepCI, + Branch: authorization.Branch, RemoteRef: ciHead}, nil + }) + if err != nil { + return summary + } + summary.CIRef = green.RemoteRef + summary.State, summary.Code, summary.Detail = PublishStatePublished, "", "" + return summary +} + +// settleCI polls one head until none of its checks is pending, and returns +// them. Unlike the CI wait it does not stop at a red check, and it does not +// follow the branch: a head that moves ends it with ci_rerun_head_moved. +func (w *Worker) settleCI( + ctx context.Context, gateway PullRequestGateway, options PublishOptions, + target publishTarget, head string, deadline time.Time, view *rerunView, +) ([]CICheck, error) { + transient := 0 + for { + current, err := w.remoteHead(ctx, target, target.branch) + if err == nil && current != head { + return nil, publishFailure(ciRerunHeadMoved, + "the branch moved from %s to %s while jig waited to re-run its failed jobs", + shortSHA(head), shortSHA(current)) + } + var checks []CICheck + if err == nil { + checks, err = gateway.CommitChecks(ctx, target.repository, head) + } + var pending []CICheck + if err != nil { + transient++ + if transient >= protocol.MaxCITransientFailures { + return nil, publishFailure("ci_unavailable", + "CI state could not be read %d times in a row: %s", + transient, boundedText(err.Error(), protocol.MaxPublishDiagnosticBytes)) + } + } else { + transient = 0 + checks = view.apply(checks) + for _, check := range checks { + if check.Verdict == CIPending { + pending = append(pending, check) + } + } + if len(pending) == 0 { + return checks, nil + } + } + if !time.Now().Before(deadline) { + return nil, publishFailure("ci_timeout", + "CI on %s did not finish before the re-run could be requested; still pending: %s", + shortSHA(head), describeChecks(pending)) + } + if err := w.ciPause(ctx, options, target, head, deadline); err != nil { + return nil, err + } + } +} + +func failedChecks(checks []CICheck) []CICheck { + var failed []CICheck + for _, check := range checks { + if check.Verdict == CIFail { + failed = append(failed, check) + } + } + return failed +} diff --git a/internal/worker/publish_rerun_test.go b/internal/worker/publish_rerun_test.go new file mode 100644 index 0000000..eac49ae --- /dev/null +++ b/internal/worker/publish_rerun_test.go @@ -0,0 +1,637 @@ +// publish_rerun_test.go — declared re-runs of flaky Actions checks (plan +// U4) through the real worker, control plane, and git remote, with CI and +// the re-run endpoint scripted by the fake gateway. +package worker + +import ( + "context" + "errors" + "fmt" + "os" + "path/filepath" + "slices" + "strings" + "testing" + "time" + + "github.com/StructuPath/jig/internal/protocol" +) + +// rerunSnapshot waits for CI with a re-run budget and, when repairBudget is +// positive, a repair loop too. +func rerunSnapshot(rerunBudget, repairBudget int) string { + if repairBudget == 0 { + return integrationSnapshot + fmt.Sprintf(`publish: + ci: + wait: true + rerun: {budget: %d} +`, rerunBudget) + } + return integrationSnapshot + fmt.Sprintf(` - name: review + kind: agent + owner: builder +publish: + ci: + wait: true + rerun: {budget: %d} + on_fail: {run: build, budget: %d} +`, rerunBudget, repairBudget) +} + +func actionsJob(name, verdict string, id int64) CICheck { + check := check(name, verdict) + check.App = "github-actions" + check.CheckRunID = id + return check +} + +// flakyCI scripts CI per head: script is called with the 0-based index of +// the head being judged (in the order heads are first seen), the number of +// re-run requests made so far, and the 1-based poll count. It runs under +// the gateway's lock, so it reads g.reruns directly. +type flakyCI struct { + heads []string + script func(head, reruns, poll int) []CICheck +} + +func (c *flakyCI) install(g *fakeGateway) { + g.checks = func(sha string, poll int) ([]CICheck, error) { + index := slices.Index(c.heads, sha) + if index < 0 { + c.heads = append(c.heads, sha) + index = len(c.heads) - 1 + } + return c.script(index, len(g.reruns), poll), nil + } +} + +func newRerunScenario(t *testing.T, rerunBudget, repairBudget int, script func(head, reruns, poll int) []CICheck, + rounds ...roundScript) (*repairScenario, *flakyCI) { + t.Helper() + s := &repairScenario{h: newHarness(t), ci: &ciScript{red: map[string]bool{}}} + var head, identity string + s.originDir, head, identity = newOriginRepo(t) + s.h.seedRunWithSnapshot("run-1", rerunSnapshot(rerunBudget, repairBudget), + protocol.RunTarget{Repository: identity, BaseSHA: head}) + s.job = s.h.enqueue("run-1", identity) + s.branch = protocol.PublishBranch(s.job.ID, 1) + s.continuation = &scriptedContinuation{t: t, rounds: rounds} + s.gateway = newFakeGateway() + ci := &flakyCI{script: script} + ci.install(s.gateway) + s.w = newCIWorker(t, s.h, repairRunner(t, s.continuation), s.gateway) + return s, ci +} + +// flakyTest is red until the first re-run, then its new run (a new check-run +// id) passes. lint is a green Actions sibling throughout. +func flakyTest(_, reruns, _ int) []CICheck { + if reruns == 0 { + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIFail, 11)} + } + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIPass, 12)} +} + +// brokenTest is red on every run, each re-run a new check-run id. +func brokenTest(head, reruns, _ int) []CICheck { + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIFail, int64(100*head+11+reruns))} +} + +func assertReruns(t *testing.T, summary PublishSummary, want ...CIRerunSummary) { + t.Helper() + if len(summary.CIReruns) != len(want) { + t.Fatalf("ci_reruns = %+v, want %d entries", summary.CIReruns, len(want)) + } + for i, entry := range want { + got := summary.CIReruns[i] + if got.Attempt != entry.Attempt || got.Outcome != entry.Outcome || + (entry.Head != "" && got.Head != entry.Head) || + (entry.Jobs != nil && !slices.Equal(got.Jobs, entry.Jobs)) || + (entry.Detail != "" && !strings.Contains(got.Detail, entry.Detail)) { + t.Fatalf("ci_reruns[%d] = %+v, want %+v", i, got, entry) + } + } +} + +// A single flaky Actions job that passes on re-run publishes green, flagged +// flaky, and no repair round runs even though one was declared (R5, R6). +func TestAFlakyActionsJobPassesOnRerunAndPublishes(t *testing.T) { + s, ci := newRerunScenario(t, 1, 1, flakyTest, fixRound()) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.Published() || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted, published, and flagged flaky", attempt.State, summary) + } + pushed := remoteBranches(t, s.originDir)[s.branch] + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Head: pushed, Jobs: []string{"test"}, Outcome: "passed"}) + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{11}) { + t.Fatalf("re-run requests = %v, want only the failed job 11", requests) + } + if len(s.continuation.failures) != 0 || len(s.rounds(t, attempt.ID)) != 0 || len(summary.CIRepairs) != 0 { + t.Fatal("a repair round ran although the re-run went green") + } + if len(ci.heads) != 1 || summary.CIRef != pushed || summary.RemoteRef != pushed { + t.Fatalf("heads judged = %v ci_ref = %s, want CI judged green on the one pushed head %s", + ci.heads, summary.CIRef, pushed) + } + if steps := recordedSteps(t, s.w, attempt.ID); !slices.Contains(steps, protocol.PublishStepCI) { + t.Fatalf("ledger steps = %v, want ci recorded after the re-run", steps) + } + if !strings.Contains(attempt.Result, `"ci_reruns":[{"attempt":1,`) || + !strings.Contains(attempt.Result, `"jobs":["test"],"outcome":"passed"`) { + t.Fatalf("result = %s, want the pinned ci_reruns shape", attempt.Result) + } +} + +// A sibling still running when the first check goes red delays the re-run +// until it finishes: GitHub refuses to re-run a job in a running workflow. +func TestARerunWaitsForAPendingSiblingToFinish(t *testing.T) { + const siblingDoneAt = 4 + var polls int + s, _ := newRerunScenario(t, 1, 0, func(_, reruns, poll int) []CICheck { + polls = poll + slow := actionsJob("slow-e2e", CIPending, 21) + if poll >= siblingDoneAt { + slow = actionsJob("slow-e2e", CIPass, 21) + } + if reruns == 0 { + return []CICheck{actionsJob("test", CIFail, 11), slow} + } + return []CICheck{actionsJob("test", CIPass, 12), slow} + }) + requestedAt := 0 + s.gateway.rerun = func(int64, int) error { + requestedAt = polls + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted after the re-run", attempt.State, summary) + } + if requestedAt < siblingDoneAt { + t.Fatalf("the re-run was requested at poll %d, before the sibling finished at poll %d", + requestedAt, siblingDoneAt) + } +} + +// The first polls after a re-run still return the old red check run: it is +// pending until a newer run of the same name appears, never read back red. +func TestTheStaleRedCheckAfterARerunIsPending(t *testing.T) { + var rerunAt int + s, _ := newRerunScenario(t, 1, 0, func(_, reruns, poll int) []CICheck { + lint := actionsJob("lint", CIPass, 1) + switch { + case reruns == 0: + return []CICheck{lint, actionsJob("test", CIFail, 11)} + case poll <= rerunAt+3: + // GitHub has not created the new run yet. + return []CICheck{lint, actionsJob("test", CIFail, 11)} + case poll <= rerunAt+5: + return []CICheck{lint, actionsJob("test", CIPending, 12)} + } + return []CICheck{lint, actionsJob("test", CIPass, 12)} + }) + s.gateway.rerun = func(int64, int) error { + rerunAt = s.gateway.checkPolls + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted { + t.Fatalf("attempt = %s summary = %+v, want the stale red run waited past", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "passed"}) + if requests := s.gateway.rerunRequests(); len(requests) != 1 { + t.Fatalf("re-run requests = %v, want one", requests) + } +} + +// A same-named check from another workflow that was already on the head is +// not the re-run's replacement: the re-run job stays pending until its own +// new run appears. +func TestAnOlderSameNamedCheckDoesNotReplaceTheRerun(t *testing.T) { + view := newRerunView([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIPass, 12)}) + view.rerun[11] = true + viewed := view.apply([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIPass, 12)}) + if len(viewed) != 2 || viewed[0].Verdict != CIPending { + t.Fatalf("viewed = %+v, want the re-run check pending beside the other workflow's", viewed) + } + viewed = view.apply([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIPass, 12), + actionsJob("build", CIPass, 13)}) + if len(viewed) != 2 || viewed[0].CheckRunID != 12 || viewed[1].CheckRunID != 13 { + t.Fatalf("viewed = %+v, want the stale run dropped once its new run exists", viewed) + } + if plain := (*rerunView)(nil).apply([]CICheck{actionsJob("build", CIFail, 11)}); plain[0].Verdict != CIFail { + t.Fatal("a nil view rewrote a check") + } +} + +// GitHub refusing because the workflow run is still in progress is waited +// out and asked again, and does not spend the budget. +func TestAnInProgressRefusalIsRetriedWithoutSpendingTheBudget(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, flakyTest) + s.gateway.rerun = func(_ int64, call int) error { + if call <= 2 { + return publishFailure(ciRerunInProgress, "This workflow is already running") + } + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted after the retried re-run", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "passed"}) + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{11, 11, 11}) { + t.Fatalf("re-run requests = %v, want job 11 asked three times", requests) + } +} + +// An "in progress" refusal that never clears is bounded: after the retries +// run out it is recorded as a refusal. +func TestAnEndlessInProgressRefusalIsBounded(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, flakyTest) + s.gateway.rerun = func(int64, int) error { + return publishFailure(ciRerunInProgress, "This workflow is already running") + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "ci_rerun_refused", Detail: "in progress"}) + if requests := s.gateway.rerunRequests(); len(requests) != maxCIRerunInProgressRetries+1 { + t.Fatalf("re-run requests = %d, want %d", len(requests), maxCIRerunInProgressRetries+1) + } +} + +// Any other refusal is recorded with its diagnostic and falls through: to a +// repair round when one is declared, to ci_failed when not. +func TestAnyOtherRefusalIsRecordedAndFallsThrough(t *testing.T) { + refuse := func(int64, int) error { + return publishFailure("gh_failed", "gh api job rerun failed: Resource not accessible by integration") + } + t.Run("to a repair round", func(t *testing.T) { + s, _ := newRerunScenario(t, 2, 1, func(head, reruns, poll int) []CICheck { + if head == 0 { + return brokenTest(head, reruns, poll) + } + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIPass, 111)} + }, fixRound()) + s.gateway.rerun = refuse + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted by the repair round, not flaky", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "ci_rerun_refused", + Detail: "Resource not accessible"}) + if len(summary.CIRepairs) != 1 || len(s.gateway.rerunRequests()) != 1 { + t.Fatalf("repairs = %+v requests = %v, want one round after one refused request", + summary.CIRepairs, s.gateway.rerunRequests()) + } + }) + t.Run("to ci_failed", func(t *testing.T) { + s, _ := newRerunScenario(t, 2, 0, brokenTest) + s.gateway.rerun = refuse + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "ci_rerun_refused"}) + if requests := s.gateway.rerunRequests(); len(requests) != 1 { + t.Fatalf("re-run requests = %v, want one: a refusal stops re-running this head", requests) + } + }) +} + +// A check still red after the budget goes on to a repair round when one is +// declared, and ends ci_failed with every re-run recorded when not. +func TestACheckStillRedAfterTheBudgetFallsThrough(t *testing.T) { + t.Run("to a repair round", func(t *testing.T) { + s, _ := newRerunScenario(t, 1, 1, func(head, reruns, poll int) []CICheck { + if head == 0 { + return brokenTest(head, reruns, poll) + } + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIPass, 111)} + }, fixRound()) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted by the round, not flaky", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "failed"}) + if len(summary.CIRepairs) != 1 || summary.CIRepairs[0].Outcome != "pushed" { + t.Fatalf("ci_repairs = %+v, want one pushed round", summary.CIRepairs) + } + if len(s.continuation.failures) != 1 || s.continuation.failures[0].Checks[0].CheckRunID != 12 { + t.Fatalf("the round was handed %+v, want the re-run's own red run", s.continuation.failures) + } + }) + t.Run("to ci_failed", func(t *testing.T) { + s, _ := newRerunScenario(t, 2, 0, brokenTest) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "failed"}, + CIRerunSummary{Attempt: 2, Outcome: "failed"}) + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{11, 12}) { + t.Fatalf("re-run requests = %v, want each red run re-run once, within the budget", requests) + } + }) +} + +// Any red check that is not an Actions job skips re-runs entirely (R7). +func TestARedNonActionsCheckSkipsReruns(t *testing.T) { + for name, red := range map[string][]CICheck{ + "commit status": {check("legacy/ci", CIFail)}, + "mixed": {actionsJob("test", CIFail, 11), check("legacy/ci", CIFail)}, + "no job id": {actionsJob("test", CIFail, 0)}, + "other app": {{Name: "codecov", Verdict: CIFail, App: "codecov", CheckRunID: 31}}, + } { + t.Run(name, func(t *testing.T) { + s, _ := newRerunScenario(t, 3, 0, func(int, int, int) []CICheck { return red }) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + if len(summary.CIReruns) != 0 || len(s.gateway.rerunRequests()) != 0 { + t.Fatalf("ci_reruns = %+v requests = %v, want none", summary.CIReruns, s.gateway.rerunRequests()) + } + }) + } +} + +// A red sibling that is not an Actions job, found only once CI settles, +// also stops the re-run before any request. +func TestARedNonActionsSiblingFoundWhileSettlingSkipsTheRerun(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, func(_, _, poll int) []CICheck { + legacy := check("legacy/ci", CIPending) + if poll >= 3 { + legacy = check("legacy/ci", CIFail) + } + return []CICheck{actionsJob("test", CIFail, 11), legacy} + }) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + if len(s.gateway.rerunRequests()) != 0 || len(summary.CIFailures) != 2 { + t.Fatalf("requests = %v failures = %+v, want no request and both red checks named", + s.gateway.rerunRequests(), summary.CIFailures) + } +} + +// The budget is per attempt: a repair round's new red head gets what the +// earlier head left, and nothing once it is spent. +func TestTheRerunBudgetCarriesAcrossRepairRounds(t *testing.T) { + t.Run("what is left applies to the new head", func(t *testing.T) { + s, ci := newRerunScenario(t, 2, 1, func(head, reruns, poll int) []CICheck { + if head == 0 { + return brokenTest(head, reruns, poll) + } + if reruns <= 1 { + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIFail, 111)} + } + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIPass, 112)} + }, fixRound()) + s.gateway.rerun = func(_ int64, call int) error { + if call == 1 { + return publishFailure("gh_failed", "refused") + } + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted, flaky on the fix", attempt.State, summary) + } + assertReruns(t, summary, + CIRerunSummary{Attempt: 1, Head: ci.heads[0], Outcome: "ci_rerun_refused"}, + CIRerunSummary{Attempt: 2, Head: ci.heads[1], Outcome: "passed"}) + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{11, 111}) { + t.Fatalf("re-run requests = %v, want one per head", requests) + } + }) + t.Run("a spent budget does not", func(t *testing.T) { + s, _ := newRerunScenario(t, 1, 1, brokenTest, fixRound()) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_repair_exhausted" { + t.Fatalf("attempt = %s summary = %+v, want ci_repair_exhausted", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "failed"}) + if requests := s.gateway.rerunRequests(); len(requests) != 1 || len(summary.CIRepairs) != 1 { + t.Fatalf("requests = %v repairs = %+v, want one re-run and one round", requests, summary.CIRepairs) + } + }) +} + +// A lease lost before a re-run means no request is sent — before each +// request, not only the first. +func TestALostLeaseSendsNoRerun(t *testing.T) { + for _, tc := range []struct { + name string + loseFrom int // the request count after which the lease is lost + want []int64 + }{ + {"before the first", 0, nil}, + {"between two jobs", 1, []int64{11}}, + } { + t.Run(tc.name, func(t *testing.T) { + w, gateway, target, pushed := publishedWorker(t) + var lost bool + // The attempt is already terminal, so a heartbeat is refused: + // a stale local clock is all it takes to lose the lease. + target.lease.now = func() time.Time { + if lost { + return time.Now().Add(time.Hour) + } + return time.Now() + } + lost = tc.loseFrom == 0 + gateway.checks = func(string, int) ([]CICheck, error) { + return []CICheck{actionsJob("test", CIFail, 11), actionsJob("lint", CIFail, 12)}, nil + } + gateway.rerun = func(_ int64, call int) error { + if call >= tc.loseFrom { + lost = true + } + return nil + } + summary := PublishSummary{State: PublishStateFailed, Code: "ci_failed", RemoteRef: pushed, CIRef: pushed, + CIFailures: []CICheck{actionsJob("test", CIFail, 11)}} + summary = w.rerunFlakyCI(context.Background(), gateway, fastCIOptions(), target, summary, + ciPolicy{wait: true, timeout: time.Minute, rerunBudget: 1}) + if requests := gateway.rerunRequests(); !slices.Equal(requests, tc.want) { + t.Fatalf("re-run requests = %v, want %v", requests, tc.want) + } + if summary.Published() || summary.Code == "ci_failed" || len(summary.CIReruns) != 1 || + summary.CIReruns[0].Outcome != summary.Code { + t.Fatalf("summary = %+v, want the publish ended on the lease verdict, recorded", summary) + } + }) + } +} + +// A person pushing while jig waits to re-run ends the re-run: there is +// nothing of jig's left on the branch to re-run. +func TestAPersonsPushWhileSettlingEndsTheRerun(t *testing.T) { + s, _ := newRerunScenario(t, 1, 1, nil, fixRound()) + var personal string + s.gateway.checks = func(_ string, poll int) ([]CICheck, error) { + if poll == 2 { + personal = pushToBranch(t, s.originDir, s.branch) + } + return []CICheck{actionsJob("test", CIFail, 11), actionsJob("slow", CIPending, 21)}, nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_rerun_head_moved" { + t.Fatalf("attempt = %s summary = %+v, want ci_rerun_head_moved", attempt.State, summary) + } + if len(s.gateway.rerunRequests()) != 0 || len(s.continuation.failures) != 0 || + remoteBranches(t, s.originDir)[s.branch] != personal { + t.Fatal("a re-run or a repair round ran on top of a person's push") + } +} + +// A publish-only retry never re-runs: like repair, it judges CI and nothing +// more. +func TestThePublishRetryNeverReruns(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, brokenTest) + if attempt, _ := s.run(t); attempt.State != protocol.AttemptAcceptedUnpublished { + t.Fatalf("attempt = %s, want accepted_unpublished", attempt.State) + } + retried, err := s.w.RetryPublish(context.Background(), s.job.ID, s.gateway, fastCIOptions()) + if err != nil { + t.Fatalf("publish retry: %v", err) + } + if summary := publishSummaryOf(t, retried.Result); summary.Code != "ci_failed" || len(summary.CIReruns) != 0 || + len(s.gateway.rerunRequests()) != 1 { + t.Fatalf("retry summary = %+v requests = %v, want red judged with no new re-run", + summary, s.gateway.rerunRequests()) + } +} + +// ---- the gh gateway ---------------------------------------------------------- + +func TestGitHubGatewayRerunsOneActionsJob(t *testing.T) { + var calls [][]string + var stdout, stderr []byte + var runErr error + gateway := &GitHubCLIGateway{ + LookPath: func(string) (string, error) { return "/usr/bin/gh", nil }, + Run: func(_ context.Context, name string, arguments ...string) ([]byte, []byte, bool, bool, error) { + calls = append(calls, append([]string{name}, arguments...)) + return stdout, stderr, false, false, runErr + }, + } + if err := gateway.RerunActionsJob(context.Background(), "github.com/acme/widgets", 4242); err != nil { + t.Fatalf("rerun: %v", err) + } + want := []string{"gh", "api", "-X", "POST", "-H", "Accept: application/vnd.github+json", + "repos/acme/widgets/actions/jobs/4242/rerun"} + if len(calls) != 1 || !slices.Equal(calls[0], want) { + t.Fatalf("gh calls = %v, want %v", calls, want) + } + + for _, tc := range []struct { + name string + stdout, stderr string + code string + }{ + {"in progress on stderr", "", "gh: This workflow is already running (HTTP 403)", ciRerunInProgress}, + {"in progress in the body", `{"message":"Cannot rerun a job while the run is in progress"}`, + "gh: HTTP 403", ciRerunInProgress}, + {"forbidden", `{"message":"Resource not accessible by integration"}`, + "gh: Resource not accessible by integration (HTTP 403)", "gh_failed"}, + {"unauthenticated", "", "gh: To get started with GitHub CLI, please run: gh auth login", "gh_unauthenticated"}, + } { + t.Run(tc.name, func(t *testing.T) { + stdout, stderr, runErr = []byte(tc.stdout), []byte(tc.stderr), errors.New("exit status 1") + err := gateway.RerunActionsJob(context.Background(), "github.com/acme/widgets", 4242) + if publishCode(err) != tc.code { + t.Fatalf("err = %v (code %s), want %s", err, publishCode(err), tc.code) + } + }) + } + + stdout, stderr, runErr = nil, nil, nil + calls = nil + if err := gateway.RerunActionsJob(context.Background(), "github.com/acme/widgets", 0); publishCode(err) != "ci_rerun_invalid_job" { + t.Fatalf("err = %v, want ci_rerun_invalid_job", err) + } + if err := gateway.RerunActionsJob(context.Background(), "gitlab.com/acme/widgets", 1); publishCode(err) != "publish_unsupported_remote" { + t.Fatalf("err = %v, want publish_unsupported_remote", err) + } + if len(calls) != 0 { + t.Fatalf("gh was called for an invalid request: %v", calls) + } +} + +// ---- the gated real-gh test -------------------------------------------------- + +// rerunGateWorkflow fails its flaky job on a workflow run's first attempt +// and passes on any re-run; its slow sibling keeps the run in progress well +// after the flaky job goes red, which is the case the fakes could not see: +// GitHub refuses to re-run a job while its workflow run is still going. +const rerunGateWorkflow = `name: jig-rerun-gate +on: + push: + branches: ['jig/**'] +jobs: + flaky: + runs-on: ubuntu-latest + steps: + - run: test "${{ github.run_attempt }}" != "1" + slow: + runs-on: ubuntu-latest + steps: + - run: sleep 90 +` + +// TestCIRerunGate re-runs a real Actions job through real gh after its slow +// sibling finishes. It is skipped unless JIG_RERUN_GATE names a scratch +// repository (owner/repo) whose other workflows, if any, pass on a jig/** +// branch. The attempt's change is the gate workflow itself, so gh needs the +// workflow scope; the test opens a pull request and leaves it for the +// operator to close. +func TestCIRerunGate(t *testing.T) { + project := strings.TrimSpace(os.Getenv("JIG_RERUN_GATE")) + if project == "" { + t.Skip("set JIG_RERUN_GATE=owner/repo to run the flaky re-run gate") + } + ctx := context.Background() + identity := "github.com/" + project + stdout, err := runGit(ctx, "", "ls-remote", "https://github.com/"+project+".git", "HEAD") + if err != nil { + t.Fatalf("read %s HEAD: %v", project, err) + } + baseSHA, _, found := strings.Cut(strings.TrimSpace(stdout), "\t") + if !found || len(baseSHA) < 40 { + t.Fatalf("unreadable HEAD for %s: %q", project, stdout) + } + + h := newHarness(t) + h.seedRunWithSnapshot("run-gate", integrationSnapshot+`publish: + ci: + wait: true + timeout: 20m + rerun: {budget: 1} +`, protocol.RunTarget{Repository: identity, BaseSHA: baseSHA}) + h.enqueue("run-gate", identity) + options := PublishOptions{CommitAuthorName: "jig-test", CommitAuthorEmail: "jig@test", + CIPollInterval: 10 * time.Second} + runner := NewPublishingRunner(writeAndDeclare(t, map[string]string{ + ".github/workflows/jig-rerun-gate.yml": rerunGateWorkflow, + }), NewGitHubCLIGateway(), options) + w := newTestWorker(t, h, filepath.Join(t.TempDir(), "worker"), 1, runner) + runner.Bind(w) + + attempt, err := w.ClaimOnce(ctx) + if err != nil || attempt == nil { + t.Fatalf("claim: attempt=%v err=%v", attempt, err) + } + summary := publishSummaryOf(t, attempt.Result) + t.Logf("rerun gate: state=%s pull request=%s summary=%+v", attempt.State, summary.PullRequestURL, summary) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted and flagged flaky", attempt.State, summary) + } + if len(summary.CIReruns) != 1 || summary.CIReruns[0].Outcome != "passed" || + !slices.Equal(summary.CIReruns[0].Jobs, []string{"flaky"}) { + t.Fatalf("ci_reruns = %+v, want the flaky job re-run once and passed", summary.CIReruns) + } +} diff --git a/internal/worker/publish_test.go b/internal/worker/publish_test.go index 3586ad2..d54d9df 100644 --- a/internal/worker/publish_test.go +++ b/internal/worker/publish_test.go @@ -39,6 +39,27 @@ type fakeGateway struct { checks func(sha string, poll int) ([]CICheck, error) checkPolls int checkedSHA []string + // rerun scripts re-run requests: it is called with the job id and the + // 1-based request count. Nil accepts every request. reruns lists the + // job ids of every request made, accepted or refused. + rerun func(jobID int64, call int) error + reruns []int64 +} + +func (g *fakeGateway) RerunActionsJob(_ context.Context, _ string, jobID int64) error { + g.mutex.Lock() + defer g.mutex.Unlock() + g.reruns = append(g.reruns, jobID) + if g.rerun == nil { + return nil + } + return g.rerun(jobID, len(g.reruns)) +} + +func (g *fakeGateway) rerunRequests() []int64 { + g.mutex.Lock() + defer g.mutex.Unlock() + return append([]int64(nil), g.reruns...) } func newFakeGateway() *fakeGateway { From d60474684e80a684fa7f936a6133e9b96095058e Mon Sep 17 00:00:00 2001 From: Victor Garcia Date: Mon, 28 Sep 2026 00:31:19 -0600 Subject: [PATCH 2/8] test(worker): pin re-run waits between refusals and no re-run on a person's head Co-Authored-By: Claude Opus 5.5 --- internal/worker/publish_rerun_test.go | 29 +++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) diff --git a/internal/worker/publish_rerun_test.go b/internal/worker/publish_rerun_test.go index eac49ae..90bc795 100644 --- a/internal/worker/publish_rerun_test.go +++ b/internal/worker/publish_rerun_test.go @@ -228,7 +228,9 @@ func TestAnOlderSameNamedCheckDoesNotReplaceTheRerun(t *testing.T) { // out and asked again, and does not spend the budget. func TestAnInProgressRefusalIsRetriedWithoutSpendingTheBudget(t *testing.T) { s, _ := newRerunScenario(t, 1, 0, flakyTest) + var pollsAt []int s.gateway.rerun = func(_ int64, call int) error { + pollsAt = append(pollsAt, s.gateway.checkPolls) if call <= 2 { return publishFailure(ciRerunInProgress, "This workflow is already running") } @@ -238,6 +240,11 @@ func TestAnInProgressRefusalIsRetriedWithoutSpendingTheBudget(t *testing.T) { if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { t.Fatalf("attempt = %s summary = %+v, want accepted after the retried re-run", attempt.State, summary) } + for i := 1; i < len(pollsAt); i++ { + if pollsAt[i] <= pollsAt[i-1] { + t.Fatalf("CI polls at each request = %v: a refusal was asked again without waiting on CI", pollsAt) + } + } assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "passed"}) if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{11, 11, 11}) { t.Fatalf("re-run requests = %v, want job 11 asked three times", requests) @@ -488,6 +495,28 @@ func TestAPersonsPushWhileSettlingEndsTheRerun(t *testing.T) { } } +// CI red on a head jig did not push (a person pushed before CI was judged) +// is theirs to fix: no re-run is requested on it. +func TestRedCIOnAPersonsHeadIsNotRerun(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, nil) + var personal string + s.gateway.checks = func(_ string, poll int) ([]CICheck, error) { + if poll == 1 { + personal = pushToBranch(t, s.originDir, s.branch) + return []CICheck{actionsJob("test", CIPending, 11)}, nil + } + return []CICheck{actionsJob("test", CIFail, 11)}, nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" || summary.CIRef != personal { + t.Fatalf("attempt = %s summary = %+v, want ci_failed on the person's head %s", attempt.State, summary, personal) + } + if len(s.gateway.rerunRequests()) != 0 || len(summary.CIReruns) != 0 { + t.Fatalf("requests = %v ci_reruns = %+v, want no re-run on a person's head", + s.gateway.rerunRequests(), summary.CIReruns) + } +} + // A publish-only retry never re-runs: like repair, it judges CI and nothing // more. func TestThePublishRetryNeverReruns(t *testing.T) { From b69aadd7523abc8c592dc54a70f9a9106797d343 Mon Sep 17 00:00:00 2001 From: Victor Garcia Date: Mon, 28 Sep 2026 01:08:11 -0600 Subject: [PATCH 3/8] test(worker): kill the replacement-count mutant and fail fast on forbidden re-runs Co-Authored-By: Claude Opus 5.5 --- internal/worker/publish_rerun_test.go | 34 ++++++++++++++++++++++++++- 1 file changed, 33 insertions(+), 1 deletion(-) diff --git a/internal/worker/publish_rerun_test.go b/internal/worker/publish_rerun_test.go index 90bc795..c55319e 100644 --- a/internal/worker/publish_rerun_test.go +++ b/internal/worker/publish_rerun_test.go @@ -97,6 +97,16 @@ func brokenTest(head, reruns, _ int) []CICheck { return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIFail, int64(100*head+11+reruns))} } +// forbidReruns fails the test on any re-run request, and refuses it so the +// run under test ends promptly instead of waiting on a re-run that never +// lands. +func forbidReruns(t *testing.T) func(int64, int) error { + return func(jobID int64, _ int) error { + t.Errorf("a re-run of job %d was requested", jobID) + return publishFailure("gh_failed", "re-runs are forbidden in this test") + } +} + func assertReruns(t *testing.T, summary PublishSummary, want ...CIRerunSummary) { t.Helper() if len(summary.CIReruns) != len(want) { @@ -219,6 +229,20 @@ func TestAnOlderSameNamedCheckDoesNotReplaceTheRerun(t *testing.T) { if len(viewed) != 2 || viewed[0].CheckRunID != 12 || viewed[1].CheckRunID != 13 { t.Fatalf("viewed = %+v, want the stale run dropped once its new run exists", viewed) } + // Two same-named jobs re-run (one per workflow): one new run is not + // both replacements, so both stay pending until the second appears. + view = newRerunView([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIFail, 12)}) + view.rerun[11], view.rerun[12] = true, true + viewed = view.apply([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIFail, 12), + actionsJob("build", CIPass, 13)}) + if len(viewed) != 3 || viewed[0].Verdict != CIPending || viewed[1].Verdict != CIPending { + t.Fatalf("viewed = %+v, want both re-run checks pending until both are replaced", viewed) + } + viewed = view.apply([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIFail, 12), + actionsJob("build", CIPass, 13), actionsJob("build", CIPass, 14)}) + if len(viewed) != 2 { + t.Fatalf("viewed = %+v, want both stale runs dropped once both new runs exist", viewed) + } if plain := (*rerunView)(nil).apply([]CICheck{actionsJob("build", CIFail, 11)}); plain[0].Verdict != CIFail { t.Fatal("a nil view rewrote a check") } @@ -353,6 +377,7 @@ func TestARedNonActionsCheckSkipsReruns(t *testing.T) { } { t.Run(name, func(t *testing.T) { s, _ := newRerunScenario(t, 3, 0, func(int, int, int) []CICheck { return red }) + s.gateway.rerun = forbidReruns(t) attempt, summary := s.run(t) if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) @@ -374,6 +399,7 @@ func TestARedNonActionsSiblingFoundWhileSettlingSkipsTheRerun(t *testing.T) { } return []CICheck{actionsJob("test", CIFail, 11), legacy} }) + s.gateway.rerun = forbidReruns(t) attempt, summary := s.run(t) if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) @@ -478,12 +504,17 @@ func TestALostLeaseSendsNoRerun(t *testing.T) { // nothing of jig's left on the branch to re-run. func TestAPersonsPushWhileSettlingEndsTheRerun(t *testing.T) { s, _ := newRerunScenario(t, 1, 1, nil, fixRound()) + s.gateway.rerun = forbidReruns(t) var personal string s.gateway.checks = func(_ string, poll int) ([]CICheck, error) { if poll == 2 { personal = pushToBranch(t, s.originDir, s.branch) } - return []CICheck{actionsJob("test", CIFail, 11), actionsJob("slow", CIPending, 21)}, nil + slow := actionsJob("slow", CIPending, 21) + if poll >= 4 { + slow = actionsJob("slow", CIPass, 21) + } + return []CICheck{actionsJob("test", CIFail, 11), slow}, nil } attempt, summary := s.run(t) if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_rerun_head_moved" { @@ -499,6 +530,7 @@ func TestAPersonsPushWhileSettlingEndsTheRerun(t *testing.T) { // is theirs to fix: no re-run is requested on it. func TestRedCIOnAPersonsHeadIsNotRerun(t *testing.T) { s, _ := newRerunScenario(t, 1, 0, nil) + s.gateway.rerun = forbidReruns(t) var personal string s.gateway.checks = func(_ string, poll int) ([]CICheck, error) { if poll == 1 { From 7a8c23164d2edcc2b67f3473dfc26165c51a4668 Mon Sep 17 00:00:00 2001 From: Victor Garcia Date: Mon, 28 Sep 2026 01:23:39 -0600 Subject: [PATCH 4/8] fix(worker): pin re-run judgement to its head, bound it, and re-run per workflow run Review fixes for the flaky re-run step (U4): - judge the head the re-run ran on, never where the branch moved; a push after the request ends ci_rerun_head_moved instead of a flaky pass - a re-run that never finishes or whose CI cannot be read (ci_timeout, ci_unavailable) is recorded and leaves CI red as it was, so repair runs - settle, requests, and judgement share one CI timeout per re-run - re-run failed jobs once per workflow run (rerun-failed-jobs), not per job, so matrix siblings are not refused while one re-runs - fence each request on cancellation and a recorded lost-lease verdict, not only freshen, which skips a fresh lease - no in-progress retry once the deadline has passed Co-Authored-By: Claude Opus 5.5 --- README.md | 21 ++- cmd/jig/ci_repair_test.go | 10 +- internal/worker/claiming.go | 16 ++ internal/worker/publish.go | 74 ++++++-- internal/worker/publish_rerun.go | 181 ++++++++++++++----- internal/worker/publish_rerun_test.go | 251 ++++++++++++++++++++++++-- internal/worker/publish_test.go | 26 ++- 7 files changed, 479 insertions(+), 100 deletions(-) diff --git a/README.md b/README.md index e65ff6e..95c8745 100644 --- a/README.md +++ b/README.md @@ -260,14 +260,21 @@ Rules worth knowing before you write one: `publish: {ci: {wait: true, rerun: {budget: 1}}}` re-runs the failed GitHub Actions jobs on the same head when CI is red on a head jig pushed, before any repair round (or before the job ends red, with no `on_fail`). - It first waits, within the CI timeout, until nothing on the head is still - running, since GitHub will not re-run a job in a running workflow; then it - re-runs each failed job with `gh api -X POST - repos/{owner}/{repo}/actions/jobs/{id}/rerun`, freshening the lease before - every request, and awaits CI again, reading each re-run job as pending - until its new run replaces the old red one. GitHub refusing because a + It first waits until nothing on the head is still running, since GitHub + will not re-run a job in a running workflow; then, for each workflow run + the failed jobs belong to, it makes one `gh api -X POST + repos/{owner}/{repo}/actions/runs/{id}/rerun-failed-jobs` request (per run, + not per job: re-running one job would put the run in progress and GitHub + would refuse its siblings), checking the lease and cancellation right + before each. It then judges that same head again, reading each re-run job + as pending until its new run replaces the old red one; if the branch moves + meanwhile, the re-run ends `ci_rerun_head_moved`, so a person's fix is + never reported as jig's flaky pass. The whole re-run — waiting, requests, + and judgement — fits in one CI timeout. GitHub refusing because a workflow run is in progress is waited out and does not spend the budget; - any other refusal is recorded and falls through to repair or the end. If + any other refusal is recorded and falls through to repair or the end. A + re-run GitHub accepted that never finishes, or whose CI cannot be read, is + recorded and leaves CI red as it was, so repair still runs. If any red check is not an Actions job (a commit status, another app), no re-run happens. Re-runs never move the branch. The budget, 1–3, is per attempt: a repair round's new head gets only what is left. Every re-run is diff --git a/cmd/jig/ci_repair_test.go b/cmd/jig/ci_repair_test.go index 4216dbb..6440968 100644 --- a/cmd/jig/ci_repair_test.go +++ b/cmd/jig/ci_repair_test.go @@ -75,9 +75,13 @@ func (g *e2eGateway) FailedCheckLogs(_ context.Context, _ string, checks []worke return checks } -// RerunActionsJob is never reached: the definition declares no re-runs, and -// its checks are not Actions jobs. -func (g *e2eGateway) RerunActionsJob(context.Context, string, int64) error { +// ActionsJobRun and RerunFailedJobs are never reached: the definition +// declares no re-runs, and its checks are not Actions jobs. +func (g *e2eGateway) ActionsJobRun(context.Context, string, int64) (int64, error) { + return 0, fmt.Errorf("unexpected workflow run lookup") +} + +func (g *e2eGateway) RerunFailedJobs(context.Context, string, int64) error { return fmt.Errorf("unexpected re-run request") } diff --git a/internal/worker/claiming.go b/internal/worker/claiming.go index 959145f..26e17bb 100644 --- a/internal/worker/claiming.go +++ b/internal/worker/claiming.go @@ -42,6 +42,17 @@ type attemptLease struct { cancelled chan struct{} cancelOnce sync.Once + + // lost is the control plane's verdict that this lease can never renew + // again, once any heartbeat has received one. + lost error +} + +// lostVerdict reports the lease-lost verdict a heartbeat received, or nil. +func (l *attemptLease) lostVerdict() error { + l.mutex.Lock() + defer l.mutex.Unlock() + return l.lost } func newAttemptLease(client *Client, attemptID, token string) *attemptLease { @@ -60,6 +71,11 @@ func newAttemptLease(client *Client, attemptID, token string) *attemptLease { func (l *attemptLease) heartbeat(ctx context.Context) error { response, err := l.client.Heartbeat(ctx, l.attemptID, protocol.HeartbeatRequest{LeaseToken: l.token}) if err != nil { + if leaseLost(err) { + l.mutex.Lock() + l.lost = err + l.mutex.Unlock() + } return err } l.mutex.Lock() diff --git a/internal/worker/publish.go b/internal/worker/publish.go index 1e6ca58..f8834c3 100644 --- a/internal/worker/publish.go +++ b/internal/worker/publish.go @@ -175,11 +175,13 @@ type PullRequestGateway interface { // job's log attached, within the CI repair log bounds. It never fails: // a log that cannot be read leaves a LogNote saying why. FailedCheckLogs(ctx context.Context, repository string, checks []CICheck) []CICheck - // RerunActionsJob asks GitHub to re-run one failed Actions job on the - // same commit. A refusal because the job's workflow run is still in - // progress carries the code ci_rerun_in_progress, so it can be waited - // out rather than counted. - RerunActionsJob(ctx context.Context, repository string, jobID int64) error + // ActionsJobRun reports the workflow run one Actions job belongs to. + ActionsJobRun(ctx context.Context, repository string, jobID int64) (int64, error) + // RerunFailedJobs asks GitHub to re-run every failed job of one + // workflow run on the same commit. A refusal because the run is still + // in progress carries the code ci_rerun_in_progress, so it can be + // waited out rather than counted. + RerunFailedJobs(ctx context.Context, repository string, runID int64) error } // CI check verdicts, normalized across check runs and commit statuses. @@ -1562,34 +1564,66 @@ func (g *GitHubCLIGateway) FailedCheckLogs(ctx context.Context, repository strin return annotated } -// RerunActionsJob re-runs one failed Actions job with -// `gh api -X POST repos/{project}/actions/jobs/{id}/rerun`: per job, on the -// same commit, never moving the branch (KTD5). GitHub refuses while the -// job's workflow run is still going; that refusal is ci_rerun_in_progress. -func (g *GitHubCLIGateway) RerunActionsJob(ctx context.Context, repository string, jobID int64) error { - project, err := githubProject(repository) +// ActionsJobRun reads the workflow run an Actions job belongs to with +// `gh api repos/{project}/actions/jobs/{id} --jq .run_id`. +func (g *GitHubCLIGateway) ActionsJobRun(ctx context.Context, repository string, jobID int64) (int64, error) { + project, err := g.rerunPreflight(repository, jobID) if err != nil { - return err + return 0, err } - if jobID <= 0 { - return publishFailure("ci_rerun_invalid_job", "%d is not an Actions job id", jobID) + stdout, stderr, stdoutTooLarge, stderrTooLarge, err := g.run()(ctx, "gh", + "api", "-H", "Accept: application/vnd.github+json", + fmt.Sprintf("repos/%s/actions/jobs/%d", project, jobID), "--jq", ".run_id") + if err != nil { + return 0, ghDiagnostic("gh api actions job", err, stderr, stdoutTooLarge, stderrTooLarge) } - if _, err := g.lookPath()("gh"); err != nil { - return publishFailure("gh_not_found", - "the GitHub CLI (gh) was not found on PATH. Install gh, then run `gh auth login`.") + run, err := strconv.ParseInt(strings.TrimSpace(string(stdout)), 10, 64) + if err != nil || run <= 0 { + return 0, publishFailure("gh_malformed_output", "gh api actions job %d reported no workflow run id", jobID) + } + return run, nil +} + +// RerunFailedJobs re-runs every failed job of one workflow run with +// `gh api -X POST repos/{project}/actions/runs/{id}/rerun-failed-jobs`: on +// the same commit, never moving the branch. It is per run rather than per +// job (the plan's KTD5) because re-running one job puts its run in +// progress, and GitHub then refuses every sibling in it. GitHub refuses +// while the run is still going; that refusal is ci_rerun_in_progress. +func (g *GitHubCLIGateway) RerunFailedJobs(ctx context.Context, repository string, runID int64) error { + project, err := g.rerunPreflight(repository, runID) + if err != nil { + return err } stdout, stderr, stdoutTooLarge, stderrTooLarge, err := g.run()(ctx, "gh", "api", "-X", "POST", "-H", "Accept: application/vnd.github+json", - fmt.Sprintf("repos/%s/actions/jobs/%d/rerun", project, jobID)) + fmt.Sprintf("repos/%s/actions/runs/%d/rerun-failed-jobs", project, runID)) if err == nil { return nil } if rerunInProgress(stdout) || rerunInProgress(stderr) { return publishFailure(ciRerunInProgress, - "GitHub will not re-run job %d while its workflow run is in progress: %s", jobID, + "GitHub will not re-run workflow run %d while it is in progress: %s", runID, boundedText(strings.TrimSpace(string(stderr)), protocol.MaxPublishDiagnosticBytes)) } - return ghDiagnostic("gh api job rerun", err, stderr, stdoutTooLarge, stderrTooLarge) + return ghDiagnostic("gh api rerun-failed-jobs", err, stderr, stdoutTooLarge, stderrTooLarge) +} + +// rerunPreflight resolves the project and refuses an id that is not one, +// before gh is run. +func (g *GitHubCLIGateway) rerunPreflight(repository string, id int64) (string, error) { + project, err := githubProject(repository) + if err != nil { + return "", err + } + if id <= 0 { + return "", publishFailure("ci_rerun_invalid_id", "%d is not an Actions job or run id", id) + } + if _, err := g.lookPath()("gh"); err != nil { + return "", publishFailure("gh_not_found", + "the GitHub CLI (gh) was not found on PATH. Install gh, then run `gh auth login`.") + } + return project, nil } // rerunInProgress recognizes GitHub's refusal to re-run a job whose diff --git a/internal/worker/publish_rerun.go b/internal/worker/publish_rerun.go index 639c586..aba8e49 100644 --- a/internal/worker/publish_rerun.go +++ b/internal/worker/publish_rerun.go @@ -6,15 +6,25 @@ // red CI → every red check an Actions job, budget left? // → wait until nothing on the head is pending (GitHub refuses to // re-run a job whose workflow run is still going) -// → freshen the lease, re-run each failed job (an "in progress" +// → group the failed jobs by workflow run; per run, fence on the +// lease and ask GitHub to re-run its failed jobs (an "in progress" // refusal waits and asks again; it spends nothing) -// → a fresh ci step on the same head, the re-run check runs pending -// until newer runs of the same name replace them +// → judge the SAME head again, the re-run check runs pending until +// newer runs of the same name replace them // → green publishes, flagged flaky; red loops while budget remains // -// A re-run never moves the branch, so it is fenced by the lease rather than -// the ledger, and every one is recorded in the summary's ci_reruns. The -// budget is per attempt: a repair round's new head gets only what is left. +// One re-run — settle, requests, and judgement — fits inside one CI +// timeout. A re-run never moves the branch, so it is fenced by the lease +// rather than the ledger, and every one is recorded in the summary's +// ci_reruns. It never turns a red CI that a repair round could take into a +// terminal stop: a re-run that GitHub accepted but that never finished is +// recorded, and CI is left red as it was. The budget is per attempt: a +// repair round's new head gets only what is left. +// +// Re-runs are per workflow run (`rerun-failed-jobs`), not per job as the +// plan's KTD5 first said: re-running one job puts its run in progress, so +// GitHub refuses every sibling in the same run until it finishes, and N +// failed matrix shards would cost N workflow durations. package worker import ( @@ -50,10 +60,12 @@ const ( // re-run; there is nothing of jig's left to re-run. ciRerunHeadMoved = "ci_rerun_head_moved" // ciRerunInProgress is the gateway's code for GitHub refusing a re-run - // because the job's workflow run has not finished. It is waited out, - // never recorded as a spent re-run. + // because the workflow run has not finished. It is waited out, never + // recorded as a spent re-run. ciRerunInProgress = "ci_rerun_in_progress" - // maxCIRerunInProgressRetries bounds how often one job's re-run is asked + // ciRerunCancelled: the job was cancelled before a re-run request. + ciRerunCancelled = "ci_rerun_cancelled" + // maxCIRerunInProgressRetries bounds how often one run's re-run is asked // for again after an "in progress" refusal; the CI timeout bounds how // long. maxCIRerunInProgressRetries = 5 @@ -132,8 +144,10 @@ func (v *rerunView) apply(checks []CICheck) []CICheck { // every red check is an Actions job, and budget remains. summary is a // publish summary that ended on ci_failed; the returned one is either // published (flaky), still ci_failed (for a repair round or the end), or -// ended on the code a re-run stopped on. Every pass through the loop records -// one ci_reruns entry or returns, so it runs at most the budget's times. +// ended on the code a re-run stopped on (a moved head, a lost lease, a +// cancellation). Every pass through the loop records one ci_reruns entry or +// returns, so it runs at most the budget's times, and each pass is bounded +// by one CI timeout. func (w *Worker) rerunFlakyCI( ctx context.Context, gateway PullRequestGateway, @@ -151,6 +165,7 @@ func (w *Worker) rerunFlakyCI( } entry := CIRerunSummary{Attempt: len(summary.CIReruns) + 1, Head: head, Jobs: checkNames(summary.CIFailures)} + // One deadline for the whole re-run: settle, requests, judgement. deadline := time.Now().Add(ci.timeout) // The CI wait stopped at the first red check; its siblings may still @@ -162,8 +177,13 @@ func (w *Worker) rerunFlakyCI( failed := failedChecks(checks) if len(failed) == 0 { // Nothing is red any more (someone re-ran it). Judge the head - // afresh; with no re-run of jig's, a pass here is not flaky. - return w.judgeAfterRerun(ctx, gateway, options, target, summary, ci, nil) + // afresh; with no re-run of jig's, a pass here is not flaky, and + // a judgement that cannot finish leaves CI red as it was. + judged := w.judgeAfterRerun(ctx, gateway, options, target, summary, head, deadline, nil) + if unjudged(judged.Code) { + return summary + } + return judged } summary.CIFailures = boundedChecks(failed) entry.Jobs = checkNames(failed) @@ -176,24 +196,40 @@ func (w *Worker) rerunFlakyCI( if err := w.requestReruns(ctx, gateway, options, target, head, deadline, summary.CIFailures, view); err != nil { return rerunStopped(summary, entry, err) } - summary = w.judgeAfterRerun(ctx, gateway, options, target, summary, ci, view) + judged := w.judgeAfterRerun(ctx, gateway, options, target, summary, head, deadline, view) switch { - case summary.Published(): - summary.CIFlaky = true + case judged.Published(): + judged.CIFlaky = true entry.Outcome = ciRerunPassed - case summary.Code == "ci_failed": + case judged.Code == "ci_failed": entry.Outcome = ciRerunFailed - entry.Detail = summary.Detail + entry.Detail = judged.Detail + case unjudged(judged.Code): + // GitHub took the re-run but it never finished (an outage, a + // queued concurrency group) or CI could not be read: that is + // no verdict on the code, so CI stays red exactly as it was and + // a repair round, or the plain red stop, takes it from here. + entry.Outcome = judged.Code + entry.Detail = judged.Detail + summary.CIReruns = append(summary.CIReruns, entry) + return summary default: - entry.Outcome = summary.Code - entry.Detail = summary.Detail + entry.Outcome = judged.Code + entry.Detail = judged.Detail } + summary = judged summary.CIReruns = append(summary.CIReruns, entry) } return summary } -// rerunStopped records a re-run that stopped before its CI wait. A lost +// unjudged reports whether a re-run's judgement ended without a verdict on +// the code: CI did not finish in time, or could not be read. +func unjudged(code string) bool { + return code == "ci_timeout" || code == "ci_unavailable" +} + +// rerunStopped records a re-run that stopped before its judgement. A lost // lease, a cancellation, and a moved head end the publish on that code; any // other stop (a refusal, CI that never settled) leaves CI red, so a repair // round or the attempt's end takes it from there. Either way the entry is @@ -203,38 +239,60 @@ func rerunStopped(summary PublishSummary, entry CIRerunSummary, err error) Publi entry.Outcome = code entry.Detail = boundedText(err.Error(), protocol.MaxPublishDiagnosticBytes) summary.CIReruns = append(summary.CIReruns, entry) - switch code { - case ciRerunRefused, "ci_timeout", "ci_unavailable": + if code == ciRerunRefused || unjudged(code) { return summary } return failSummary(summary, code, err.Error()) } -// requestReruns asks GitHub to re-run each failed job, freshening the lease -// before every request so an attempt that lost its lease sends none. An "in -// progress" refusal is waited out and asked again, within the deadline and -// maxCIRerunInProgressRetries; any other refusal stops the re-run. +// requestReruns asks GitHub to re-run the failed jobs of each workflow run +// they belong to, once per run. Every request is fenced: the lease is +// freshened and checked for cancellation or loss immediately before it, so +// an attempt that lost its lease or was cancelled sends none. An "in +// progress" refusal is waited out and asked again, while the deadline +// allows and at most maxCIRerunInProgressRetries times; any other refusal +// stops the re-run. func (w *Worker) requestReruns( ctx context.Context, gateway PullRequestGateway, options PublishOptions, target publishTarget, head string, deadline time.Time, failed []CICheck, view *rerunView, ) error { + runs, order := map[int64][]CICheck{}, []int64{} for _, job := range failed { + run, err := gateway.ActionsJobRun(ctx, target.repository, job.CheckRunID) + if err != nil { + return publishFailure(ciRerunRefused, "the workflow run of %s could not be read: %s", job.Name, err.Error()) + } + if _, seen := runs[run]; !seen { + order = append(order, run) + } + runs[run] = append(runs[run], job) + } + for _, run := range order { + jobs := runs[run] for retries := 0; ; retries++ { - if err := target.lease.freshen(ctx); err != nil { - return fmt.Errorf("lease could not be freshened: %w", err) + if err := fenceRerun(ctx, target); err != nil { + return err } - err := gateway.RerunActionsJob(ctx, target.repository, job.CheckRunID) + err := gateway.RerunFailedJobs(ctx, target.repository, run) if err == nil { - view.rerun[job.CheckRunID] = true + for _, job := range jobs { + view.rerun[job.CheckRunID] = true + } break } if publishCode(err) != ciRerunInProgress { - return publishFailure(ciRerunRefused, "GitHub refused to re-run %s: %s", job.Name, err.Error()) + return publishFailure(ciRerunRefused, "GitHub refused to re-run %s: %s", + describeChecks(jobs), err.Error()) } if retries == maxCIRerunInProgressRetries { return publishFailure(ciRerunRefused, - "GitHub still reported %s's workflow run in progress after %d waits: %s", - job.Name, retries, err.Error()) + "GitHub still reported the workflow run of %s in progress after %d waits: %s", + describeChecks(jobs), retries, err.Error()) + } + if !time.Now().Before(deadline) { + return publishFailure(ciRerunRefused, + "GitHub still reported the workflow run of %s in progress at the CI timeout: %s", + describeChecks(jobs), err.Error()) } if err := w.ciPause(ctx, options, target, head, deadline); err != nil { return err @@ -247,24 +305,55 @@ func (w *Worker) requestReruns( return nil } -// judgeAfterRerun is a fresh, fenced ci step on the branch: green is -// recorded and publishes, anything else ends the step as the CI wait does. +// fenceRerun is the lease fence before one re-run request: heartbeat if the +// lease may be stale, then refuse if the lease is known lost or the job was +// cancelled — a heartbeat reports cancellation without failing, and a fresh +// lease is not heartbeated at all, so both are checked here. +func fenceRerun(ctx context.Context, target publishTarget) error { + if err := target.lease.freshen(ctx); err != nil { + return fmt.Errorf("lease could not be freshened: %w", err) + } + if err := target.lease.lostVerdict(); err != nil { + return fmt.Errorf("the lease was lost: %w", err) + } + select { + case <-target.lease.cancelled: + return publishFailure(ciRerunCancelled, "the job was cancelled before its failed jobs were re-run") + default: + return nil + } +} + +// judgeAfterRerun judges the head the re-run ran on — never whatever the +// branch has moved to — within the re-run's deadline, and records a green +// verdict through the fenced ci step. A branch that moves after the re-run +// ends it with ci_rerun_head_moved: a person's push is theirs to judge, not +// jig's flaky pass. func (w *Worker) judgeAfterRerun( ctx context.Context, gateway PullRequestGateway, options PublishOptions, - target publishTarget, summary PublishSummary, ci ciPolicy, view *rerunView, + target publishTarget, summary PublishSummary, head string, deadline time.Time, view *rerunView, ) PublishSummary { + summary.CIRef = head + checks, err := w.settleCI(ctx, gateway, options, target, head, deadline, view) + if err == nil && len(checks) == 0 { + err = publishFailure("ci_unavailable", "no CI checks were reported on %s after the re-run", shortSHA(head)) + } + if err == nil { + if failed := failedChecks(checks); len(failed) > 0 { + summary.CIFailures = boundedChecks(failed) + err = publishFailure("ci_failed", "%d CI check(s) failed on %s after the re-run: %s", + len(failed), shortSHA(head), describeChecks(failed)) + } + } + if err != nil { + return failSummary(summary, publishCode(err), + fmt.Sprintf("publish step %q: %s", protocol.PublishStepCI, err.Error())) + } summary.CIFailures = nil green, err := w.publishStep(ctx, target, protocol.PublishStepCI, map[string]protocol.PublishRecord{}, &summary, func(authorization protocol.PublishAuthorization) (protocol.PublishStepRequest, error) { - ciHead, failures, err := w.awaitCIViewed(ctx, gateway, options, target, authorization.Branch, - ci.timeout, view) - summary.CIRef = ciHead - summary.CIFailures = failures - if err != nil { - return protocol.PublishStepRequest{}, err - } return protocol.PublishStepRequest{Step: protocol.PublishStepCI, - Branch: authorization.Branch, RemoteRef: ciHead}, nil + Branch: authorization.Branch, RemoteRef: head}, nil }) if err != nil { return summary @@ -315,7 +404,7 @@ func (w *Worker) settleCI( } if !time.Now().Before(deadline) { return nil, publishFailure("ci_timeout", - "CI on %s did not finish before the re-run could be requested; still pending: %s", + "CI on %s did not finish within the re-run's CI timeout; still pending: %s", shortSHA(head), describeChecks(pending)) } if err := w.ciPause(ctx, options, target, head, deadline); err != nil { diff --git a/internal/worker/publish_rerun_test.go b/internal/worker/publish_rerun_test.go index c55319e..7f260d6 100644 --- a/internal/worker/publish_rerun_test.go +++ b/internal/worker/publish_rerun_test.go @@ -567,9 +567,195 @@ func TestThePublishRetryNeverReruns(t *testing.T) { } } +// ---- review fixes -------------------------------------------------------------- + +// A person pushing after jig's re-run request is theirs to judge: the +// judgement stays pinned to the re-run head and ends ci_rerun_head_moved, +// never recording the person's green head as jig's flaky pass. +func TestAPersonsPushAfterTheRerunIsNotAFlakyPass(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, func(head, reruns, _ int) []CICheck { + if head == 0 && reruns == 0 { + return []CICheck{actionsJob("test", CIFail, 11)} + } + if head == 0 { + return []CICheck{actionsJob("test", CIPending, 12)} + } + return []CICheck{actionsJob("test", CIPass, 21)} + }) + var personal string + s.gateway.rerun = func(int64, int) error { + personal = pushToBranch(t, s.originDir, s.branch) + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_rerun_head_moved" || + summary.CIFlaky || summary.Published() { + t.Fatalf("attempt = %s summary = %+v, want ci_rerun_head_moved and not flaky", attempt.State, summary) + } + if len(summary.CIReruns) != 1 || summary.CIReruns[0].Outcome != "ci_rerun_head_moved" || + summary.CIReruns[0].Head == personal || summary.CIRef == personal { + t.Fatalf("summary = %+v, want the re-run recorded on jig's head %s, not the person's", summary, personal) + } + if steps := recordedSteps(t, s.w, attempt.ID); slices.Contains(steps, protocol.PublishStepCI) { + t.Fatalf("ledger steps = %v: CI was recorded green on a head jig did not re-run", steps) + } +} + +// A re-run whose CI cannot be read afterwards is no verdict on the code: CI +// stays red as it was, and the declared repair round still runs. +func TestAnUnreadableJudgementAfterARerunStillReachesRepair(t *testing.T) { + s, _ := newRerunScenario(t, 1, 1, nil, fixRound()) + ci := &flakyCI{} + s.gateway.checks = func(sha string, _ int) ([]CICheck, error) { + index := slices.Index(ci.heads, sha) + if index < 0 { + ci.heads = append(ci.heads, sha) + index = len(ci.heads) - 1 + } + switch { + case index > 0: + return []CICheck{actionsJob("test", CIPass, 111)}, nil + case len(s.gateway.reruns) == 0: + return []CICheck{actionsJob("test", CIFail, 11)}, nil + } + return nil, errors.New("connection reset") + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted by the repair round", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Head: ci.heads[0], Outcome: "ci_unavailable"}) + if len(summary.CIRepairs) != 1 || len(s.continuation.failures) != 1 || + s.continuation.failures[0].Checks[0].CheckRunID != 11 { + t.Fatalf("repairs = %+v handed = %+v, want one round handed the original red check", + summary.CIRepairs, s.continuation.failures) + } +} + +// One re-run — settle, request, and judgement — fits in one CI timeout, and +// a re-run GitHub accepted but never started leaves CI red as it was. +func TestARerunThatNeverFinishesIsBoundedAndLeavesCIRed(t *testing.T) { + w, gateway, target, pushed := publishedWorker(t) + start := time.Now() + const timeout = 400 * time.Millisecond + gateway.checks = func(string, int) ([]CICheck, error) { + slow := actionsJob("slow", CIPending, 21) + if time.Since(start) > timeout*3/4 { + slow = actionsJob("slow", CIPass, 21) + } + // The re-run is accepted, but its new run never appears. + return []CICheck{actionsJob("test", CIFail, 11), slow}, nil + } + red := []CICheck{actionsJob("test", CIFail, 11)} + summary := PublishSummary{State: PublishStateFailed, Code: "ci_failed", Detail: "1 CI check(s) failed", + RemoteRef: pushed, CIRef: pushed, CIFailures: red} + summary = w.rerunFlakyCI(context.Background(), gateway, fastCIOptions(), target, summary, + ciPolicy{wait: true, timeout: timeout, rerunBudget: 3}) + elapsed := time.Since(start) + if elapsed > timeout+timeout/2 { + t.Fatalf("the re-run took %s, more than one CI timeout of %s", elapsed, timeout) + } + if summary.Code != "ci_failed" || summary.CIRef != pushed || len(summary.CIFailures) != 1 || + summary.CIFailures[0].CheckRunID != 11 || summary.Detail != "1 CI check(s) failed" { + t.Fatalf("summary = %+v, want CI left red exactly as it was", summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Head: pushed, Outcome: "ci_timeout"}) + if requests := gateway.rerunRequests(); !slices.Equal(requests, []int64{11}) { + t.Fatalf("re-run requests = %v, want one, and no second re-run of a run still going", requests) + } +} + +// Failed jobs of one workflow run are re-run with one request for the run, +// not one per job (whose siblings GitHub would refuse while it runs). +func TestFailedJobsOfOneWorkflowRunAreRerunTogether(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, func(_, reruns, _ int) []CICheck { + if reruns == 0 { + return []CICheck{actionsJob("test (1)", CIFail, 11), actionsJob("test (2)", CIFail, 12), + actionsJob("lint", CIFail, 31)} + } + return []CICheck{actionsJob("test (1)", CIPass, 13), actionsJob("test (2)", CIPass, 14), + actionsJob("lint", CIPass, 32)} + }) + s.gateway.runOf = func(job int64) int64 { + if job == 31 { + return 3000 + } + return 1000 + } + s.gateway.rerun = func(run int64, _ int) error { + if run != 1000 && run != 3000 { + t.Errorf("re-run of run %d, want a workflow run id", run) + } + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted and flaky", attempt.State, summary) + } + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{1000, 3000}) { + t.Fatalf("re-run requests = %v, want one per workflow run", requests) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "passed", + Jobs: []string{"test (1)", "test (2)", "lint"}}) +} + +// A job cancelled, or a lease the control plane already declared lost, sends +// no re-run even while the lease is fresh and freshen does not heartbeat. +func TestACancelledJobOrALostVerdictSendsNoRerunOnAFreshLease(t *testing.T) { + for _, tc := range []struct { + name string + setup func(*attemptLease) + code string + }{ + {"cancelled", func(l *attemptLease) { l.cancelOnce.Do(func() { close(l.cancelled) }) }, ciRerunCancelled}, + {"lost", func(l *attemptLease) { l.lost = errors.New("lease_not_owner") }, "publish_failed"}, + } { + t.Run(tc.name, func(t *testing.T) { + w, gateway, target, pushed := publishedWorker(t) + gateway.checks = func(string, int) ([]CICheck, error) { + return []CICheck{actionsJob("test", CIFail, 11)}, nil + } + tc.setup(target.lease) + summary := PublishSummary{State: PublishStateFailed, Code: "ci_failed", RemoteRef: pushed, CIRef: pushed, + CIFailures: []CICheck{actionsJob("test", CIFail, 11)}} + summary = w.rerunFlakyCI(context.Background(), gateway, fastCIOptions(), target, summary, + ciPolicy{wait: true, timeout: time.Minute, rerunBudget: 1}) + if requests := gateway.rerunRequests(); len(requests) != 0 { + t.Fatalf("re-run requests = %v, want none", requests) + } + if summary.Code != tc.code || len(summary.CIReruns) != 1 || summary.CIReruns[0].Outcome != tc.code { + t.Fatalf("summary = %+v, want it ended on %s, recorded", summary, tc.code) + } + }) + } +} + +// Past the deadline an "in progress" refusal is not asked again: no burst of +// back-to-back requests. +func TestNoInProgressRetryPastTheDeadline(t *testing.T) { + w, gateway, target, pushed := publishedWorker(t) + gateway.checks = func(string, int) ([]CICheck, error) { + return []CICheck{actionsJob("test", CIFail, 11)}, nil + } + gateway.rerun = func(int64, int) error { + return publishFailure(ciRerunInProgress, "This workflow is already running") + } + failed := []CICheck{actionsJob("test", CIFail, 11)} + err := w.requestReruns(context.Background(), gateway, fastCIOptions(), target, pushed, + time.Now().Add(-time.Second), failed, newRerunView(failed)) + if publishCode(err) != ciRerunRefused || !strings.Contains(err.Error(), "CI timeout") { + t.Fatalf("err = %v, want a refusal at the CI timeout", err) + } + if requests := gateway.rerunRequests(); len(requests) != 1 { + t.Fatalf("re-run requests = %v, want exactly one past the deadline", requests) + } +} + // ---- the gh gateway ---------------------------------------------------------- -func TestGitHubGatewayRerunsOneActionsJob(t *testing.T) { +// The real gateway finds a job's workflow run, then re-runs the run's failed +// jobs in one request, and tells an "in progress" refusal from the rest. +func TestGitHubGatewayRerunsAWorkflowRunsFailedJobs(t *testing.T) { var calls [][]string var stdout, stderr []byte var runErr error @@ -580,11 +766,29 @@ func TestGitHubGatewayRerunsOneActionsJob(t *testing.T) { return stdout, stderr, false, false, runErr }, } - if err := gateway.RerunActionsJob(context.Background(), "github.com/acme/widgets", 4242); err != nil { + stdout = []byte("987654\n") + run, err := gateway.ActionsJobRun(context.Background(), "github.com/acme/widgets", 4242) + if err != nil || run != 987654 { + t.Fatalf("run = %d err = %v, want 987654", run, err) + } + want := []string{"gh", "api", "-H", "Accept: application/vnd.github+json", + "repos/acme/widgets/actions/jobs/4242", "--jq", ".run_id"} + if len(calls) != 1 || !slices.Equal(calls[0], want) { + t.Fatalf("gh calls = %v, want %v", calls, want) + } + for _, malformed := range []string{"", "null", "0", "abc"} { + stdout = []byte(malformed) + if _, err := gateway.ActionsJobRun(context.Background(), "github.com/acme/widgets", 4242); publishCode(err) != "gh_malformed_output" { + t.Fatalf("run id %q: err = %v, want gh_malformed_output", malformed, err) + } + } + + stdout, calls = nil, nil + if err := gateway.RerunFailedJobs(context.Background(), "github.com/acme/widgets", 987654); err != nil { t.Fatalf("rerun: %v", err) } - want := []string{"gh", "api", "-X", "POST", "-H", "Accept: application/vnd.github+json", - "repos/acme/widgets/actions/jobs/4242/rerun"} + want = []string{"gh", "api", "-X", "POST", "-H", "Accept: application/vnd.github+json", + "repos/acme/widgets/actions/runs/987654/rerun-failed-jobs"} if len(calls) != 1 || !slices.Equal(calls[0], want) { t.Fatalf("gh calls = %v, want %v", calls, want) } @@ -595,7 +799,7 @@ func TestGitHubGatewayRerunsOneActionsJob(t *testing.T) { code string }{ {"in progress on stderr", "", "gh: This workflow is already running (HTTP 403)", ciRerunInProgress}, - {"in progress in the body", `{"message":"Cannot rerun a job while the run is in progress"}`, + {"in progress in the body", `{"message":"Cannot rerun a workflow run that is in progress"}`, "gh: HTTP 403", ciRerunInProgress}, {"forbidden", `{"message":"Resource not accessible by integration"}`, "gh: Resource not accessible by integration (HTTP 403)", "gh_failed"}, @@ -603,7 +807,7 @@ func TestGitHubGatewayRerunsOneActionsJob(t *testing.T) { } { t.Run(tc.name, func(t *testing.T) { stdout, stderr, runErr = []byte(tc.stdout), []byte(tc.stderr), errors.New("exit status 1") - err := gateway.RerunActionsJob(context.Background(), "github.com/acme/widgets", 4242) + err := gateway.RerunFailedJobs(context.Background(), "github.com/acme/widgets", 987654) if publishCode(err) != tc.code { t.Fatalf("err = %v (code %s), want %s", err, publishCode(err), tc.code) } @@ -612,10 +816,13 @@ func TestGitHubGatewayRerunsOneActionsJob(t *testing.T) { stdout, stderr, runErr = nil, nil, nil calls = nil - if err := gateway.RerunActionsJob(context.Background(), "github.com/acme/widgets", 0); publishCode(err) != "ci_rerun_invalid_job" { - t.Fatalf("err = %v, want ci_rerun_invalid_job", err) + if err := gateway.RerunFailedJobs(context.Background(), "github.com/acme/widgets", 0); publishCode(err) != "ci_rerun_invalid_id" { + t.Fatalf("err = %v, want ci_rerun_invalid_id", err) } - if err := gateway.RerunActionsJob(context.Background(), "gitlab.com/acme/widgets", 1); publishCode(err) != "publish_unsupported_remote" { + if _, err := gateway.ActionsJobRun(context.Background(), "github.com/acme/widgets", -1); publishCode(err) != "ci_rerun_invalid_id" { + t.Fatalf("err = %v, want ci_rerun_invalid_id", err) + } + if err := gateway.RerunFailedJobs(context.Background(), "gitlab.com/acme/widgets", 1); publishCode(err) != "publish_unsupported_remote" { t.Fatalf("err = %v, want publish_unsupported_remote", err) } if len(calls) != 0 { @@ -625,10 +832,12 @@ func TestGitHubGatewayRerunsOneActionsJob(t *testing.T) { // ---- the gated real-gh test -------------------------------------------------- -// rerunGateWorkflow fails its flaky job on a workflow run's first attempt -// and passes on any re-run; its slow sibling keeps the run in progress well -// after the flaky job goes red, which is the case the fakes could not see: -// GitHub refuses to re-run a job while its workflow run is still going. +// rerunGateWorkflow fails both shards of its flaky matrix job on a workflow +// run's first attempt and passes on any re-run; its slow sibling keeps the +// run in progress well after the shards go red, which is the case the fakes +// could not see: GitHub refuses to re-run a job while its workflow run is +// still going. Two shards in one run prove the re-run is one request per +// run: per job, the second shard would be refused while the first re-ran. const rerunGateWorkflow = `name: jig-rerun-gate on: push: @@ -636,6 +845,10 @@ on: jobs: flaky: runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + shard: [1, 2] steps: - run: test "${{ github.run_attempt }}" != "1" slow: @@ -644,8 +857,8 @@ jobs: - run: sleep 90 ` -// TestCIRerunGate re-runs a real Actions job through real gh after its slow -// sibling finishes. It is skipped unless JIG_RERUN_GATE names a scratch +// TestCIRerunGate re-runs a real workflow run's failed jobs through real gh +// after its slow sibling finishes. It is skipped unless JIG_RERUN_GATE names a scratch // repository (owner/repo) whose other workflows, if any, pass on a jig/** // branch. The attempt's change is the gate workflow itself, so gh needs the // workflow scope; the test opens a pull request and leaves it for the @@ -691,8 +904,12 @@ func TestCIRerunGate(t *testing.T) { if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { t.Fatalf("attempt = %s summary = %+v, want accepted and flagged flaky", attempt.State, summary) } + jobs := []string(nil) + if len(summary.CIReruns) == 1 { + jobs = slices.Sorted(slices.Values(summary.CIReruns[0].Jobs)) + } if len(summary.CIReruns) != 1 || summary.CIReruns[0].Outcome != "passed" || - !slices.Equal(summary.CIReruns[0].Jobs, []string{"flaky"}) { - t.Fatalf("ci_reruns = %+v, want the flaky job re-run once and passed", summary.CIReruns) + !slices.Equal(jobs, []string{"flaky (1)", "flaky (2)"}) { + t.Fatalf("ci_reruns = %+v, want both flaky shards re-run once and passed", summary.CIReruns) } } diff --git a/internal/worker/publish_test.go b/internal/worker/publish_test.go index d54d9df..f7fbecb 100644 --- a/internal/worker/publish_test.go +++ b/internal/worker/publish_test.go @@ -39,21 +39,33 @@ type fakeGateway struct { checks func(sha string, poll int) ([]CICheck, error) checkPolls int checkedSHA []string - // rerun scripts re-run requests: it is called with the job id and the - // 1-based request count. Nil accepts every request. reruns lists the - // job ids of every request made, accepted or refused. - rerun func(jobID int64, call int) error + // runOf maps an Actions job to its workflow run; nil puts every job in a + // run of its own whose id is the job id. + runOf func(jobID int64) int64 + // rerun scripts re-run requests: it is called with the workflow run id + // and the 1-based request count. Nil accepts every request. reruns + // lists the run ids of every request made, accepted or refused. + rerun func(runID int64, call int) error reruns []int64 } -func (g *fakeGateway) RerunActionsJob(_ context.Context, _ string, jobID int64) error { +func (g *fakeGateway) ActionsJobRun(_ context.Context, _ string, jobID int64) (int64, error) { g.mutex.Lock() defer g.mutex.Unlock() - g.reruns = append(g.reruns, jobID) + if g.runOf == nil { + return jobID, nil + } + return g.runOf(jobID), nil +} + +func (g *fakeGateway) RerunFailedJobs(_ context.Context, _ string, runID int64) error { + g.mutex.Lock() + defer g.mutex.Unlock() + g.reruns = append(g.reruns, runID) if g.rerun == nil { return nil } - return g.rerun(jobID, len(g.reruns)) + return g.rerun(runID, len(g.reruns)) } func (g *fakeGateway) rerunRequests() []int64 { From 555c377bcb768813765b8863d3c7c2c2ab57671a Mon Sep 17 00:00:00 2001 From: Victor Garcia Date: Mon, 28 Sep 2026 01:24:12 -0600 Subject: [PATCH 5/8] test(worker): pin the heartbeat's recorded lost-lease verdict Co-Authored-By: Claude Opus 5.5 --- internal/worker/publish_rerun_test.go | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/internal/worker/publish_rerun_test.go b/internal/worker/publish_rerun_test.go index 7f260d6..8e23841 100644 --- a/internal/worker/publish_rerun_test.go +++ b/internal/worker/publish_rerun_test.go @@ -730,6 +730,29 @@ func TestACancelledJobOrALostVerdictSendsNoRerunOnAFreshLease(t *testing.T) { } } +// A heartbeat that receives a lost-lease verdict records it, so the next +// re-run fence refuses even though the lease now looks fresh. +func TestAHeartbeatRecordsTheLostVerdict(t *testing.T) { + w, _, target, _ := publishedWorker(t) + // A well-formed token that is not the attempt's: the control plane's + // answer is a verdict, not a malformed request. + target.lease = newAttemptLease(w.client, target.attemptID, strings.Repeat("a", 64)) + if target.lease.lostVerdict() != nil { + t.Fatal("a new lease reports a lost verdict") + } + target.lease.now = func() time.Time { return time.Now().Add(time.Hour) } + if err := target.lease.freshen(context.Background()); err == nil || !leaseLost(err) { + t.Fatalf("freshen = %v, want a lost-lease verdict from the terminal attempt", err) + } + target.lease.now = time.Now + if target.lease.lostVerdict() == nil { + t.Fatal("the heartbeat's lost verdict was not recorded") + } + if err := fenceRerun(context.Background(), target); err == nil { + t.Fatal("the re-run fence passed a lease the control plane declared lost") + } +} + // Past the deadline an "in progress" refusal is not asked again: no burst of // back-to-back requests. func TestNoInProgressRetryPastTheDeadline(t *testing.T) { From 1ea76395c513dcfcb9eead1c1a7ca20c6e657563 Mon Sep 17 00:00:00 2001 From: Victor Garcia Date: Mon, 28 Sep 2026 01:54:24 -0600 Subject: [PATCH 6/8] test(worker): kill the review-fix survivors and make the moved-head mutants fail fast Co-Authored-By: Claude Opus 5.5 --- internal/worker/publish_rerun_test.go | 28 ++++++++++++++++++++++++--- 1 file changed, 25 insertions(+), 3 deletions(-) diff --git a/internal/worker/publish_rerun_test.go b/internal/worker/publish_rerun_test.go index 8e23841..58dafc2 100644 --- a/internal/worker/publish_rerun_test.go +++ b/internal/worker/publish_rerun_test.go @@ -578,7 +578,9 @@ func TestAPersonsPushAfterTheRerunIsNotAFlakyPass(t *testing.T) { return []CICheck{actionsJob("test", CIFail, 11)} } if head == 0 { - return []CICheck{actionsJob("test", CIPending, 12)} + // jig's head would pass too: only the moved branch may stop + // it from being recorded as a flaky pass. + return []CICheck{actionsJob("test", CIPass, 12)} } return []CICheck{actionsJob("test", CIPass, 21)} }) @@ -665,11 +667,30 @@ func TestARerunThatNeverFinishesIsBoundedAndLeavesCIRed(t *testing.T) { } } +// GitHub reporting no checks at all after a re-run is no green verdict: CI +// stays red as it was. +func TestNoChecksAfterARerunIsNotGreen(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, func(_, reruns, _ int) []CICheck { + if reruns == 0 { + return []CICheck{actionsJob("test", CIFail, 11)} + } + return nil + }) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want ci_failed, not a flaky pass", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "ci_unavailable"}) +} + // Failed jobs of one workflow run are re-run with one request for the run, // not one per job (whose siblings GitHub would refuse while it runs). func TestFailedJobsOfOneWorkflowRunAreRerunTogether(t *testing.T) { - s, _ := newRerunScenario(t, 1, 0, func(_, reruns, _ int) []CICheck { - if reruns == 0 { + var lastRequestAt int + s, _ := newRerunScenario(t, 1, 0, func(_, reruns, poll int) []CICheck { + if reruns < 2 || poll <= lastRequestAt+2 { + // Before the re-runs, and for a while after: GitHub still + // lists the old red runs of every re-run job. return []CICheck{actionsJob("test (1)", CIFail, 11), actionsJob("test (2)", CIFail, 12), actionsJob("lint", CIFail, 31)} } @@ -686,6 +707,7 @@ func TestFailedJobsOfOneWorkflowRunAreRerunTogether(t *testing.T) { if run != 1000 && run != 3000 { t.Errorf("re-run of run %d, want a workflow run id", run) } + lastRequestAt = s.gateway.checkPolls return nil } attempt, summary := s.run(t) From 328f7e7e33a77026831b3e63cf306f81996cdef8 Mon Sep 17 00:00:00 2001 From: Victor Garcia Date: Mon, 28 Sep 2026 01:56:08 -0600 Subject: [PATCH 7/8] test(worker): stagger a run's new jobs so every re-run job must be read pending Co-Authored-By: Claude Opus 5.5 --- internal/worker/publish_rerun_test.go | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/internal/worker/publish_rerun_test.go b/internal/worker/publish_rerun_test.go index 58dafc2..7e9b816 100644 --- a/internal/worker/publish_rerun_test.go +++ b/internal/worker/publish_rerun_test.go @@ -694,6 +694,12 @@ func TestFailedJobsOfOneWorkflowRunAreRerunTogether(t *testing.T) { return []CICheck{actionsJob("test (1)", CIFail, 11), actionsJob("test (2)", CIFail, 12), actionsJob("lint", CIFail, 31)} } + if poll <= lastRequestAt+4 { + // The run's new jobs appear one at a time: test (2) still + // shows its old red run, which is pending, not red. + return []CICheck{actionsJob("test (1)", CIPass, 13), actionsJob("test (2)", CIFail, 12), + actionsJob("lint", CIPass, 32)} + } return []CICheck{actionsJob("test (1)", CIPass, 13), actionsJob("test (2)", CIPass, 14), actionsJob("lint", CIPass, 32)} }) From b5fe8841d3a502e20a488610bc3f67bb867ea294 Mon Sep 17 00:00:00 2001 From: Victor Garcia Date: Mon, 28 Sep 2026 01:58:35 -0600 Subject: [PATCH 8/8] fix: repair the merge's stale test header, and align the runbook with per-run re-runs The merge from main left a fragment of the old CI repair test's header in front of main's refactored helpers, so cmd/jig did not compile. The runbook (U7) described the plan's per-job re-run and a factory.yaml without re-runs; it now matches what this PR ships. Co-Authored-By: Claude Opus 5.5 --- cmd/jig/ci_repair_test.go | 3 --- docs/dogfooding.md | 43 +++++++++++++++++++++++---------------- 2 files changed, 26 insertions(+), 20 deletions(-) diff --git a/cmd/jig/ci_repair_test.go b/cmd/jig/ci_repair_test.go index bc60836..dfcf257 100644 --- a/cmd/jig/ci_repair_test.go +++ b/cmd/jig/ci_repair_test.go @@ -93,9 +93,6 @@ func (g *e2eGateway) RerunFailedJobs(context.Context, string, int64) error { return fmt.Errorf("unexpected re-run request") } -func TestACIRepairRoundRunsThroughTheRealWorkerWiring(t *testing.T) { - repo := initRepo(t) - serverData, workerData := t.TempDir(), t.TempDir() // scriptCIRepairRuntime scripts the chain's build and the repair round's. func scriptCIRepairRuntime(t *testing.T) { t.Helper() diff --git a/docs/dogfooding.md b/docs/dogfooding.md index 500aa32..b0c59da 100644 --- a/docs/dogfooding.md +++ b/docs/dogfooding.md @@ -21,11 +21,13 @@ The report is only as good as the ledger underneath it. Run the trial on a | `jig report` and `GET /api/report` (PR #24) | there is no report. `jig report --help` must print its usage. | | Spend on every agent phase exit (PR #25) | spend omits every phase that did not pass. It also overcounts every resumed session, because Claude Code's `total_cost_usd` is a running total per session. | | Publish history across retries (PR #22) | a publish-only retry erases the stop code it replaces, so the *retried, still red* rate reads zero. | -| Flaky-check re-runs (`publish.ci.rerun`, U4 of the plan) | optional. Without it, the *Re-runs* section is all zeros and section 3's masking threshold does not apply. | +| Flaky-check re-runs (`publish.ci.rerun`, PR #28) | optional. Without it, the *Re-runs* section is all zeros and section 3's masking threshold does not apply. | -This runbook describes re-runs as the plan specifies them (R5–R7, KTD5). The -PR that implements them had not landed when this was written, so check the -README's `publish.ci` bullets for the syntax as it shipped. +This runbook describes re-runs as they shipped (R5–R7). One change from the +plan's KTD5: jig re-runs the failed jobs of each workflow run with one +`rerun-failed-jobs` request per run, not one request per job, because +re-running one job puts its run in progress and GitHub then refuses its +siblings. The README's `publish.ci` bullets have the full behaviour. Build once in your jig clone, run every command below from that clone, and use the same binary for `serve`, `worker`, and `report`: @@ -94,15 +96,22 @@ JOB=$(gh run view "$RUN" --repo "$REPO" --json jobs --jq '.jobs[0].databaseId') gh api --allow-escape-sequences -H "Accept: application/vnd.github+json" \ "repos/$REPO/actions/jobs/$JOB/logs" | tail -c 2000 -# The per-job re-run (KTD5). It really re-runs that job: CI minutes are -# spent and the job's secrets are exercised, which is exactly what jig will do. +# The job-to-run lookup jig makes before a re-run. It must print $RUN. +gh api -H "Accept: application/vnd.github+json" "repos/$REPO/actions/jobs/$JOB" --jq .run_id + +# A re-run, which needs the same Actions write permission as the +# rerun-failed-jobs request jig sends per workflow run. It really re-runs +# the job: CI minutes are spent and its secrets are exercised, which is +# exactly what jig will do. (rerun-failed-jobs itself needs a run with a +# failed job, which a green main does not have.) gh api -X POST "repos/$REPO/actions/jobs/$JOB/rerun" ``` -The first call must print log text, not an error. The second must exit 0. -If it fails with HTTP 403, the token cannot re-run jobs: fix it before -declaring any `rerun` policy. Both calls matter because the two CI repair -gate bugs showed up only against real `gh`. The fakes never saw them. +The first call must print log text, not an error. The second must print the +run id. The third must exit 0. If it fails with HTTP 403, the token cannot +re-run Actions jobs: fix it before declaring any `rerun` policy. These calls +matter because the two CI repair gate bugs showed up only against real +`gh`. The fakes never saw them. ### 1.3 CI is green on `main` @@ -169,17 +178,17 @@ Copy `factory.yaml` and edit what its header says to edit (the test command, the builder's `writes` allowlist, and the risk classifier's paths), then save it: -To re-run failed Actions jobs before a repair round is spent (requires U4), -the copy's `publish` block gains one line, a re-run budget of 1 to 3. A pass -after a re-run publishes, and the report counts it as flaky, never as a -clean first pass: +The stock `factory.yaml` re-runs failed Actions jobs once before a repair +round is spent: its `publish` block declares a re-run budget (1 to 3). A +pass after a re-run publishes, and the report counts it as flaky, never as a +clean first pass. Delete the `rerun` line to trial without re-runs: ```yaml publish: hold_when: "risk != low" ci: wait: true - rerun: {budget: 1} # the added line + rerun: {budget: 1} # delete to disable re-runs on_fail: {run: build, budget: 2} timeout: 30m ``` @@ -250,8 +259,8 @@ The worst case is roughly: |---|---|---| | `T_chain` | plan → build → test → reviewers → risk, until the PR is opened. Measure it on your first jobs (the UI lane's timeline). A round re-runs everything from `build` on, so it costs about as much. | measure | | `K` | `publish.ci.on_fail.budget`, the repair rounds | 2 | -| `R` | `publish.ci.rerun.budget`, the re-runs per attempt (0 without U4) | 1 once U4 lands | -| `T_ci` | `publish.ci.timeout`. A re-run first waits for every pending check on the head, so a slow sibling job counts against it too. | 30m | +| `R` | `publish.ci.rerun.budget`, the re-runs per attempt (0 without a `rerun` line) | 1 | +| `T_ci` | `publish.ci.timeout`. One re-run (waiting for every pending check on the head, the requests, and the judgement) fits in one `T_ci`, so a slow sibling job counts against it too. | 30m | With the stock values (K=2, R=1, T_ci=30m), the CI waits alone can take 2 hours, which leaves about 40 minutes for each of the three chain runs. If