diff --git a/README.md b/README.md index 88eaa89..2c74fc0 100644 --- a/README.md +++ b/README.md @@ -259,6 +259,35 @@ Rules worth knowing before you write one: code is the risk to design for: `factory.yaml` scores edits to CI or lint configuration, and test files that lose more lines than they gain, as not low, so such a "fix" is held for a person. +- **A flaky Actions job can be re-run before a round is spent.** + `publish: {ci: {wait: true, rerun: {budget: 1}}}` re-runs the failed + GitHub Actions jobs on the same head when CI is red on a head jig pushed, + before any repair round (or before the job ends red, with no `on_fail`). + It first waits until nothing on the head is still running, since GitHub + will not re-run a job in a running workflow; then, for each workflow run + the failed jobs belong to, it makes one `gh api -X POST + repos/{owner}/{repo}/actions/runs/{id}/rerun-failed-jobs` request (per run, + not per job: re-running one job would put the run in progress and GitHub + would refuse its siblings), checking the lease and cancellation right + before each. It then judges that same head again, reading each re-run job + as pending until its new run replaces the old red one; if the branch moves + meanwhile, the re-run ends `ci_rerun_head_moved`, so a person's fix is + never reported as jig's flaky pass. The whole re-run — waiting, requests, + and judgement — fits in one CI timeout. GitHub refusing because a + workflow run is in progress is waited out and does not spend the budget; + any other refusal is recorded and falls through to repair or the end. A + re-run GitHub accepted that never finishes, or whose CI cannot be read, is + recorded and leaves CI red as it was, so repair still runs. If + any red check is not an Actions job (a commit status, another app), no + re-run happens. Re-runs never move the branch. The budget, 1–3, is per + attempt: a repair round's new head gets only what is left. Every re-run is + in the result's `ci_reruns` (`attempt`, `head`, `jobs`, `outcome`), and a + pass after one sets `ci_flaky` — it is reported as flaky, never as a clean + pass. Re-run waits run on wall-clock time, so time spent on them is time a + later repair round no longer has under the attempt's ceiling. The + publish-only retry never re-runs. `gh` needs permission to + re-run Actions jobs on the repository. `factory.yaml` and + `factory-parallel.yaml` both declare `rerun: {budget: 1}`. - **A read-only review panel can run in parallel.** `parallel: [review-correctness, review-security, review-maintainability]` runs those phases at once, as one step of the chain. It is opt-in and narrow: one @@ -274,8 +303,9 @@ Rules worth knowing before you write one: and charges only its own budget, and then the whole group runs again, so earlier approvals are re-judged. The group runs at most 1 + the sum of its members' budgets times. `examples/definitions/factory-parallel.yaml` is the - stock factory with its panel grouped; `factory.yaml` itself stays - sequential until the parallel panel has been watched on real work. + stock factory with its panel grouped — the same phases, roles, and + `publish` block, CI re-runs and repair included; `factory.yaml` itself + stays sequential until the parallel panel has been watched on real work. - **Validation happens before anything runs.** `jig def validate ` is the same check the store applies at save time, offline. diff --git a/cmd/jig/ci_repair_test.go b/cmd/jig/ci_repair_test.go index 2dbe998..dfcf257 100644 --- a/cmd/jig/ci_repair_test.go +++ b/cmd/jig/ci_repair_test.go @@ -83,6 +83,16 @@ func (g *e2eGateway) FailedCheckLogs(_ context.Context, _ string, checks []worke return checks } +// ActionsJobRun and RerunFailedJobs are never reached: the definition +// declares no re-runs, and its checks are not Actions jobs. +func (g *e2eGateway) ActionsJobRun(context.Context, string, int64) (int64, error) { + return 0, fmt.Errorf("unexpected workflow run lookup") +} + +func (g *e2eGateway) RerunFailedJobs(context.Context, string, int64) error { + return fmt.Errorf("unexpected re-run request") +} + // scriptCIRepairRuntime scripts the chain's build and the repair round's. func scriptCIRepairRuntime(t *testing.T) { t.Helper() diff --git a/cmd/jig/factory_test.go b/cmd/jig/factory_test.go index 12b61e2..7dfdcf7 100644 --- a/cmd/jig/factory_test.go +++ b/cmd/jig/factory_test.go @@ -142,6 +142,10 @@ func TestTheFactoryRepairsRedCIThroughItsWholeReviewPanel(t *testing.T) { spec.Publish.CI.OnFail.Budget < 1 { t.Fatalf("publish.ci = %+v, want on_fail repairing through build", spec.Publish.CI) } + // One re-run tells a flaky job from a broken one before a round is spent. + if spec.Publish.CI.Rerun == nil || spec.Publish.CI.Rerun.Budget != 1 { + t.Fatalf("publish.ci.rerun = %+v, want budget 1", spec.Publish.CI.Rerun) + } after := map[string]bool{} seen := false for _, phase := range spec.Phases { diff --git a/docs/dogfooding.md b/docs/dogfooding.md index 500aa32..b0c59da 100644 --- a/docs/dogfooding.md +++ b/docs/dogfooding.md @@ -21,11 +21,13 @@ The report is only as good as the ledger underneath it. Run the trial on a | `jig report` and `GET /api/report` (PR #24) | there is no report. `jig report --help` must print its usage. | | Spend on every agent phase exit (PR #25) | spend omits every phase that did not pass. It also overcounts every resumed session, because Claude Code's `total_cost_usd` is a running total per session. | | Publish history across retries (PR #22) | a publish-only retry erases the stop code it replaces, so the *retried, still red* rate reads zero. | -| Flaky-check re-runs (`publish.ci.rerun`, U4 of the plan) | optional. Without it, the *Re-runs* section is all zeros and section 3's masking threshold does not apply. | +| Flaky-check re-runs (`publish.ci.rerun`, PR #28) | optional. Without it, the *Re-runs* section is all zeros and section 3's masking threshold does not apply. | -This runbook describes re-runs as the plan specifies them (R5–R7, KTD5). The -PR that implements them had not landed when this was written, so check the -README's `publish.ci` bullets for the syntax as it shipped. +This runbook describes re-runs as they shipped (R5–R7). One change from the +plan's KTD5: jig re-runs the failed jobs of each workflow run with one +`rerun-failed-jobs` request per run, not one request per job, because +re-running one job puts its run in progress and GitHub then refuses its +siblings. The README's `publish.ci` bullets have the full behaviour. Build once in your jig clone, run every command below from that clone, and use the same binary for `serve`, `worker`, and `report`: @@ -94,15 +96,22 @@ JOB=$(gh run view "$RUN" --repo "$REPO" --json jobs --jq '.jobs[0].databaseId') gh api --allow-escape-sequences -H "Accept: application/vnd.github+json" \ "repos/$REPO/actions/jobs/$JOB/logs" | tail -c 2000 -# The per-job re-run (KTD5). It really re-runs that job: CI minutes are -# spent and the job's secrets are exercised, which is exactly what jig will do. +# The job-to-run lookup jig makes before a re-run. It must print $RUN. +gh api -H "Accept: application/vnd.github+json" "repos/$REPO/actions/jobs/$JOB" --jq .run_id + +# A re-run, which needs the same Actions write permission as the +# rerun-failed-jobs request jig sends per workflow run. It really re-runs +# the job: CI minutes are spent and its secrets are exercised, which is +# exactly what jig will do. (rerun-failed-jobs itself needs a run with a +# failed job, which a green main does not have.) gh api -X POST "repos/$REPO/actions/jobs/$JOB/rerun" ``` -The first call must print log text, not an error. The second must exit 0. -If it fails with HTTP 403, the token cannot re-run jobs: fix it before -declaring any `rerun` policy. Both calls matter because the two CI repair -gate bugs showed up only against real `gh`. The fakes never saw them. +The first call must print log text, not an error. The second must print the +run id. The third must exit 0. If it fails with HTTP 403, the token cannot +re-run Actions jobs: fix it before declaring any `rerun` policy. These calls +matter because the two CI repair gate bugs showed up only against real +`gh`. The fakes never saw them. ### 1.3 CI is green on `main` @@ -169,17 +178,17 @@ Copy `factory.yaml` and edit what its header says to edit (the test command, the builder's `writes` allowlist, and the risk classifier's paths), then save it: -To re-run failed Actions jobs before a repair round is spent (requires U4), -the copy's `publish` block gains one line, a re-run budget of 1 to 3. A pass -after a re-run publishes, and the report counts it as flaky, never as a -clean first pass: +The stock `factory.yaml` re-runs failed Actions jobs once before a repair +round is spent: its `publish` block declares a re-run budget (1 to 3). A +pass after a re-run publishes, and the report counts it as flaky, never as a +clean first pass. Delete the `rerun` line to trial without re-runs: ```yaml publish: hold_when: "risk != low" ci: wait: true - rerun: {budget: 1} # the added line + rerun: {budget: 1} # delete to disable re-runs on_fail: {run: build, budget: 2} timeout: 30m ``` @@ -250,8 +259,8 @@ The worst case is roughly: |---|---|---| | `T_chain` | plan → build → test → reviewers → risk, until the PR is opened. Measure it on your first jobs (the UI lane's timeline). A round re-runs everything from `build` on, so it costs about as much. | measure | | `K` | `publish.ci.on_fail.budget`, the repair rounds | 2 | -| `R` | `publish.ci.rerun.budget`, the re-runs per attempt (0 without U4) | 1 once U4 lands | -| `T_ci` | `publish.ci.timeout`. A re-run first waits for every pending check on the head, so a slow sibling job counts against it too. | 30m | +| `R` | `publish.ci.rerun.budget`, the re-runs per attempt (0 without a `rerun` line) | 1 | +| `T_ci` | `publish.ci.timeout`. One re-run (waiting for every pending check on the head, the requests, and the judgement) fits in one `T_ci`, so a slow sibling job counts against it too. | 30m | With the stock values (K=2, R=1, T_ci=30m), the CI waits alone can take 2 hours, which leaves about 40 minutes for each of the three chain runs. If diff --git a/examples/definitions/factory-parallel.yaml b/examples/definitions/factory-parallel.yaml index 2df74b4..87f635f 100644 --- a/examples/definitions/factory-parallel.yaml +++ b/examples/definitions/factory-parallel.yaml @@ -58,6 +58,10 @@ # accepted_unpublished, as red CI did before; the publish retry judges CI # but never repairs. # +# THE CI RE-RUN. `rerun: {budget: 1}` re-runs the failed GitHub Actions jobs +# once, on the same head, before a repair round is spent: a job that passes +# on re-run publishes, reported as flaky; one still red goes to repair. +# # THE ROSTER. Builders and reviewers are `opus`; the planner and the extra # high-risk reviewer are `sonnet`. `effort` is medium for building to a plan # and high for reviewers hunting what the build missed. `budget_usd` caps what @@ -400,5 +404,6 @@ publish: hold_when: "risk != low" ci: wait: true + rerun: {budget: 1} on_fail: {run: build, budget: 2} timeout: 30m diff --git a/examples/definitions/factory.yaml b/examples/definitions/factory.yaml index 25e562f..361fa42 100644 --- a/examples/definitions/factory.yaml +++ b/examples/definitions/factory.yaml @@ -46,6 +46,10 @@ # accepted_unpublished, as red CI did before; the publish retry judges CI # but never repairs. # +# THE CI RE-RUN. `rerun: {budget: 1}` re-runs the failed GitHub Actions jobs +# once, on the same head, before a repair round is spent: a job that passes +# on re-run publishes, reported as flaky; one still red goes to repair. +# # THE ROSTER. Builders and reviewers are `opus`; the planner and the extra # high-risk reviewer are `sonnet`. `effort` is medium for building to a plan # and high for reviewers hunting what the build missed. `budget_usd` caps what @@ -387,5 +391,6 @@ publish: hold_when: "risk != low" ci: wait: true + rerun: {budget: 1} on_fail: {run: build, budget: 2} timeout: 30m diff --git a/internal/protocol/definition.go b/internal/protocol/definition.go index c5ec808..ddea7c4 100644 --- a/internal/protocol/definition.go +++ b/internal/protocol/definition.go @@ -136,11 +136,24 @@ type PublishSpec struct { // CISpec opts a definition into waiting for the pull request's CI after // publish. Timeout is a Go duration ("30m"); empty means DefaultCITimeout. // OnFail, when declared, repairs red CI inside the attempt instead of ending -// it there. +// it there. Rerun, when declared, re-runs failed GitHub Actions jobs on the +// same head before any repair round, to tell a flaky check from a real one. type CISpec struct { Wait bool `yaml:"wait"` Timeout string `yaml:"timeout"` OnFail *CIRepairSpec `yaml:"on_fail"` + Rerun *CIRerunSpec `yaml:"rerun"` +} + +// CIRerunSpec is the flaky-check re-run policy: +// +// rerun: {budget: N} +// +// Budget counts re-runs per attempt, not per head: a repair round's new head +// gets only what the earlier heads left. A re-run never changes the branch, +// and a pass after one is reported as flaky, never as a clean pass. +type CIRerunSpec struct { + Budget int `yaml:"budget"` } // CIRepairSpec is the CI repair loop: @@ -168,6 +181,10 @@ const ( // every phase after the repair phase, reviewers included, so a // larger budget is mostly a larger bill for a fix that is not converging. MaxCIRepairRounds = 3 + + // MaxCIReruns caps publish.ci.rerun's budget. Re-running more often than + // this stops telling a flaky check from a broken one and starts hiding it. + MaxCIReruns = 3 ) // WaitsForCI reports whether the definition gates acceptance on CI. @@ -456,9 +473,26 @@ func (spec *DefinitionSpec) validatePublish(phases map[string]PhaseSpec) error { timeout, MinCITimeout, MaxCITimeout) } } + if err := spec.validateCIRerun(); err != nil { + return err + } return spec.validateCIRepair(phases) } +func (spec *DefinitionSpec) validateCIRerun() error { + ci := spec.Publish.CI + if ci == nil || ci.Rerun == nil { + return nil + } + if !ci.Wait { + return fmt.Errorf("publish: ci: rerun re-runs red CI checks, which needs wait: true") + } + if ci.Rerun.Budget < 1 || ci.Rerun.Budget > MaxCIReruns { + return fmt.Errorf("publish: ci: rerun: budget %d is outside 1..%d", ci.Rerun.Budget, MaxCIReruns) + } + return nil +} + func (spec *DefinitionSpec) validateCIRepair(phases map[string]PhaseSpec) error { ci := spec.Publish.CI if ci == nil || ci.OnFail == nil { diff --git a/internal/protocol/definition_test.go b/internal/protocol/definition_test.go index cabaf07..c512adf 100644 --- a/internal/protocol/definition_test.go +++ b/internal/protocol/definition_test.go @@ -525,6 +525,36 @@ phases: "resume_from") } +func TestPublishCIRerunIsBoundedAndNeedsAWait(t *testing.T) { + const base = ` +name: ci-rerun +roster: + builder: {model: opus, system_prompt: s, user_prompt: u} +phases: + - {name: build, kind: agent, owner: builder} +` + if spec := mustParse(t, base+"publish: {ci: {wait: true}}\n"); spec.Publish.CI.Rerun != nil { + t.Fatalf("rerun = %+v, want nil when undeclared", spec.Publish.CI.Rerun) + } + for budget := 1; budget <= MaxCIReruns; budget++ { + spec := mustParse(t, base+fmt.Sprintf("publish: {ci: {wait: true, rerun: {budget: %d}}}\n", budget)) + if spec.Publish.CI.Rerun == nil || spec.Publish.CI.Rerun.Budget != budget { + t.Fatalf("rerun = %+v, want budget %d", spec.Publish.CI.Rerun, budget) + } + } + if MaxCIReruns != 3 { + t.Fatalf("MaxCIReruns = %d, want 3", MaxCIReruns) + } + mustReject(t, base+"publish: {ci: {wait: false, rerun: {budget: 1}}}\n", + "publish: ci: rerun", "wait: true") + mustReject(t, base+"publish: {ci: {rerun: {budget: 1}}}\n", + "publish: ci: rerun", "wait: true") + mustReject(t, base+"publish: {ci: {wait: true, rerun: {}}}\n", + "publish: ci: rerun: budget 0", "outside 1..3") + mustReject(t, base+"publish: {ci: {wait: true, rerun: {budget: 4}}}\n", + "publish: ci: rerun: budget 4", "outside 1..3") +} + func TestNegativeRoleBudgetIsRejected(t *testing.T) { spec := mustParse(t, ` name: budget @@ -579,6 +609,24 @@ func TestAParallelGroupOfConsecutiveReadOnlyReviewersValidates(t *testing.T) { } } +// A parallel panel and the CI re-run and repair policies are independent: +// declared together they validate, and each keeps its own rules. +func TestAParallelGroupValidatesWithCIRerunsAndRepair(t *testing.T) { + const publish = "publish: {ci: {wait: true, rerun: {budget: 2}, on_fail: {run: build, budget: 1}}}\n" + spec := mustParse(t, parallelPanel("", panelPhases, "parallel: [review-a, review-b, review-c]\n"+publish)) + if _, _, ok := spec.ParallelRange(); !ok { + t.Fatal("the group was lost next to a publish block") + } + if spec.Publish.CI.Rerun == nil || spec.Publish.CI.Rerun.Budget != 2 || spec.Publish.CI.OnFail == nil { + t.Fatalf("publish.ci = %+v, want the re-run and repair policies kept", spec.Publish.CI) + } + mustReject(t, parallelPanel("", panelPhases, + "parallel: [review-a, review-b, review-c]\npublish: {ci: {wait: true, rerun: {budget: 4}}}\n"), + "publish: ci: rerun: budget 4") + mustReject(t, parallelPanel("", panelPhases, + "parallel: [build, review-a]\n"+publish), "parallel") +} + // R8: every rule that keeps a group's members concurrent-safe is enforced at // save time, naming what broke it. func TestAParallelGroupIsRejectedUnlessItsMembersAreConcurrentSafe(t *testing.T) { diff --git a/internal/worker/claiming.go b/internal/worker/claiming.go index 959145f..26e17bb 100644 --- a/internal/worker/claiming.go +++ b/internal/worker/claiming.go @@ -42,6 +42,17 @@ type attemptLease struct { cancelled chan struct{} cancelOnce sync.Once + + // lost is the control plane's verdict that this lease can never renew + // again, once any heartbeat has received one. + lost error +} + +// lostVerdict reports the lease-lost verdict a heartbeat received, or nil. +func (l *attemptLease) lostVerdict() error { + l.mutex.Lock() + defer l.mutex.Unlock() + return l.lost } func newAttemptLease(client *Client, attemptID, token string) *attemptLease { @@ -60,6 +71,11 @@ func newAttemptLease(client *Client, attemptID, token string) *attemptLease { func (l *attemptLease) heartbeat(ctx context.Context) error { response, err := l.client.Heartbeat(ctx, l.attemptID, protocol.HeartbeatRequest{LeaseToken: l.token}) if err != nil { + if leaseLost(err) { + l.mutex.Lock() + l.lost = err + l.mutex.Unlock() + } return err } l.mutex.Lock() diff --git a/internal/worker/publish.go b/internal/worker/publish.go index 7bf1498..e3c81fa 100644 --- a/internal/worker/publish.go +++ b/internal/worker/publish.go @@ -127,8 +127,12 @@ type PublishSummary struct { CIFailures []CICheck `json:"ci_failures,omitempty"` // CIRepairs lists the CI repair rounds this publish ran, in order. CIRepairs []CIRepairSummary `json:"ci_repairs,omitempty"` - Performed []string `json:"performed_steps,omitempty"` - Reused []string `json:"reused_steps,omitempty"` + // CIReruns lists the flaky-check re-runs this publish ran, in order; + // CIFlaky says CI went green only after one (R6), never a clean pass. + CIReruns []CIRerunSummary `json:"ci_reruns,omitempty"` + CIFlaky bool `json:"ci_flaky,omitempty"` + Performed []string `json:"performed_steps,omitempty"` + Reused []string `json:"reused_steps,omitempty"` } // Published reports whether the pipeline completed with remote proof — the @@ -171,6 +175,13 @@ type PullRequestGateway interface { // job's log attached, within the CI repair log bounds. It never fails: // a log that cannot be read leaves a LogNote saying why. FailedCheckLogs(ctx context.Context, repository string, checks []CICheck) []CICheck + // ActionsJobRun reports the workflow run one Actions job belongs to. + ActionsJobRun(ctx context.Context, repository string, jobID int64) (int64, error) + // RerunFailedJobs asks GitHub to re-run every failed job of one + // workflow run on the same commit. A refusal because the run is still + // in progress carries the code ci_rerun_in_progress, so it can be + // waited out rather than counted. + RerunFailedJobs(ctx context.Context, repository string, runID int64) error } // CI check verdicts, normalized across check runs and commit statuses. @@ -240,6 +251,9 @@ type ciPolicy struct { // repairBudget is publish.ci.on_fail's budget; zero means red CI ends // the attempt, as it always has. repairBudget int + // rerunBudget is publish.ci.rerun's budget: flaky-check re-runs per + // attempt; zero means none. + rerunBudget int } // ciPolicyFor reads the policy out of a run snapshot. A snapshot that does @@ -255,6 +269,9 @@ func ciPolicyFor(snapshot string) ciPolicy { if policy.wait && spec.Publish.CI.OnFail != nil { policy.repairBudget = spec.Publish.CI.OnFail.Budget } + if policy.wait && spec.Publish.CI.Rerun != nil { + policy.rerunBudget = spec.Publish.CI.Rerun.Budget + } return policy } @@ -361,6 +378,9 @@ func (r *PublishingRunner) Run(ctx context.Context, prepared *PreparedAttempt) O ci := ciPolicyFor(prepared.Claim.Snapshot) summary := worker.publish(context.WithoutCancel(ctx), r.gateway, r.options, target, changedPathsFromResult(outcome.Result), ci) + // A flaky Actions job gets its declared re-runs before red CI costs a + // repair round or ends the attempt (R5). Re-runs never move the branch. + summary = worker.rerunFlakyCI(context.WithoutCancel(ctx), r.gateway, r.options, target, summary, ci) if summary.Code == "ci_failed" && outcome.Continuation != nil && ci.repairBudget > 0 { // Rounds run on the same uncancelled context: cancellation reaches // the engine through the attempt's cancel channel and the CI wait @@ -731,6 +751,16 @@ func (w *Worker) remoteHead(ctx context.Context, target publishTarget, branch st func (w *Worker) awaitCI( ctx context.Context, gateway PullRequestGateway, options PublishOptions, target publishTarget, branch string, timeout time.Duration, +) (string, []CICheck, error) { + return w.awaitCIViewed(ctx, gateway, options, target, branch, timeout, nil) +} + +// awaitCIViewed is awaitCI reading each poll through a re-run view, so a +// check run jig just re-ran counts as pending until its new run replaces it +// rather than being read back red (KTD5). A nil view is the plain wait. +func (w *Worker) awaitCIViewed( + ctx context.Context, gateway PullRequestGateway, options PublishOptions, + target publishTarget, branch string, timeout time.Duration, view *rerunView, ) (string, []CICheck, error) { start := time.Now() deadline := start.Add(timeout) @@ -761,6 +791,7 @@ func (w *Worker) awaitCI( } } else { transient = 0 + checks = view.apply(checks) var failed []CICheck pending = pending[:0] for _, check := range checks { @@ -790,24 +821,34 @@ func (w *Worker) awaitCI( return head, boundedChecks(pending), publishFailure("ci_timeout", "CI did not finish on %s within %s; %s", shortSHA(head), timeout, waiting) } - wait := options.ciPollInterval() - if remaining := time.Until(deadline); remaining < wait { - wait = remaining - } - timer := time.NewTimer(wait) - select { - case <-target.lease.cancelled: - timer.Stop() - return head, nil, publishFailure("ci_wait_cancelled", - "the job was cancelled while waiting for CI on %s", shortSHA(head)) - case <-ctx.Done(): - timer.Stop() - return head, nil, publishFailure("ci_wait_cancelled", "the CI wait was interrupted: %s", ctx.Err()) - case <-timer.C: + if err := w.ciPause(ctx, options, target, head, deadline); err != nil { + return head, nil, err } } } +// ciPause sleeps one poll interval, or until the deadline if that is +// sooner, and ends early with ci_wait_cancelled when the job is cancelled. +func (w *Worker) ciPause( + ctx context.Context, options PublishOptions, target publishTarget, head string, deadline time.Time, +) error { + wait := options.ciPollInterval() + if remaining := time.Until(deadline); remaining < wait { + wait = remaining + } + timer := time.NewTimer(wait) + defer timer.Stop() + select { + case <-target.lease.cancelled: + return publishFailure("ci_wait_cancelled", + "the job was cancelled while waiting for CI on %s", shortSHA(head)) + case <-ctx.Done(): + return publishFailure("ci_wait_cancelled", "the CI wait was interrupted: %s", ctx.Err()) + case <-timer.C: + return nil + } +} + // maxReportedCIChecks bounds how many checks a summary names. const maxReportedCIChecks = 20 @@ -1552,6 +1593,81 @@ func (g *GitHubCLIGateway) FailedCheckLogs(ctx context.Context, repository strin return annotated } +// ActionsJobRun reads the workflow run an Actions job belongs to with +// `gh api repos/{project}/actions/jobs/{id} --jq .run_id`. +func (g *GitHubCLIGateway) ActionsJobRun(ctx context.Context, repository string, jobID int64) (int64, error) { + project, err := g.rerunPreflight(repository, jobID) + if err != nil { + return 0, err + } + stdout, stderr, stdoutTooLarge, stderrTooLarge, err := g.run()(ctx, "gh", + "api", "-H", "Accept: application/vnd.github+json", + fmt.Sprintf("repos/%s/actions/jobs/%d", project, jobID), "--jq", ".run_id") + if err != nil { + return 0, ghDiagnostic("gh api actions job", err, stderr, stdoutTooLarge, stderrTooLarge) + } + run, err := strconv.ParseInt(strings.TrimSpace(string(stdout)), 10, 64) + if err != nil || run <= 0 { + return 0, publishFailure("gh_malformed_output", "gh api actions job %d reported no workflow run id", jobID) + } + return run, nil +} + +// RerunFailedJobs re-runs every failed job of one workflow run with +// `gh api -X POST repos/{project}/actions/runs/{id}/rerun-failed-jobs`: on +// the same commit, never moving the branch. It is per run rather than per +// job (the plan's KTD5) because re-running one job puts its run in +// progress, and GitHub then refuses every sibling in it. GitHub refuses +// while the run is still going; that refusal is ci_rerun_in_progress. +func (g *GitHubCLIGateway) RerunFailedJobs(ctx context.Context, repository string, runID int64) error { + project, err := g.rerunPreflight(repository, runID) + if err != nil { + return err + } + stdout, stderr, stdoutTooLarge, stderrTooLarge, err := g.run()(ctx, "gh", + "api", "-X", "POST", "-H", "Accept: application/vnd.github+json", + fmt.Sprintf("repos/%s/actions/runs/%d/rerun-failed-jobs", project, runID)) + if err == nil { + return nil + } + if rerunInProgress(stdout) || rerunInProgress(stderr) { + return publishFailure(ciRerunInProgress, + "GitHub will not re-run workflow run %d while it is in progress: %s", runID, + boundedText(strings.TrimSpace(string(stderr)), protocol.MaxPublishDiagnosticBytes)) + } + return ghDiagnostic("gh api rerun-failed-jobs", err, stderr, stdoutTooLarge, stderrTooLarge) +} + +// rerunPreflight resolves the project and refuses an id that is not one, +// before gh is run. +func (g *GitHubCLIGateway) rerunPreflight(repository string, id int64) (string, error) { + project, err := githubProject(repository) + if err != nil { + return "", err + } + if id <= 0 { + return "", publishFailure("ci_rerun_invalid_id", "%d is not an Actions job or run id", id) + } + if _, err := g.lookPath()("gh"); err != nil { + return "", publishFailure("gh_not_found", + "the GitHub CLI (gh) was not found on PATH. Install gh, then run `gh auth login`.") + } + return project, nil +} + +// rerunInProgress recognizes GitHub's refusal to re-run a job whose +// workflow run has not completed. gh prints the API message on stderr and +// the response body on stdout; either may carry it. +func rerunInProgress(output []byte) bool { + lower := strings.ToLower(string(output)) + for _, phrase := range []string{"already running", "in progress", "is running", "not complete"} { + if strings.Contains(lower, phrase) { + return true + } + } + return false +} + // maxCILogReads bounds the log reads one FailedCheckLogs call attempts, // failures included: twice the kept logs leaves room for expired or // unreadable ones without letting a hanging gh cost more than diff --git a/internal/worker/publish_repair.go b/internal/worker/publish_repair.go index 79abae3..bb57a72 100644 --- a/internal/worker/publish_repair.go +++ b/internal/worker/publish_repair.go @@ -168,6 +168,9 @@ func (w *Worker) repairCI( summary.State, summary.Code, summary.Detail = PublishStatePublished, "", "" return summary } + // Red on the round's own head: whatever re-run budget the earlier + // heads left applies here before the next round is spent (R5). + summary = w.rerunFlakyCI(ctx, gateway, options, target, summary, ci) } return summary } diff --git a/internal/worker/publish_rerun.go b/internal/worker/publish_rerun.go new file mode 100644 index 0000000..aba8e49 --- /dev/null +++ b/internal/worker/publish_rerun.go @@ -0,0 +1,424 @@ +// publish_rerun.go — declared re-runs of flaky GitHub Actions checks +// (publish.ci.rerun, plan U4, KTD5). When CI is red on the head jig pushed +// and every red check is an Actions job, jig re-runs those jobs on the same +// head before it spends a repair round or ends the attempt: +// +// red CI → every red check an Actions job, budget left? +// → wait until nothing on the head is pending (GitHub refuses to +// re-run a job whose workflow run is still going) +// → group the failed jobs by workflow run; per run, fence on the +// lease and ask GitHub to re-run its failed jobs (an "in progress" +// refusal waits and asks again; it spends nothing) +// → judge the SAME head again, the re-run check runs pending until +// newer runs of the same name replace them +// → green publishes, flagged flaky; red loops while budget remains +// +// One re-run — settle, requests, and judgement — fits inside one CI +// timeout. A re-run never moves the branch, so it is fenced by the lease +// rather than the ledger, and every one is recorded in the summary's +// ci_reruns. It never turns a red CI that a repair round could take into a +// terminal stop: a re-run that GitHub accepted but that never finished is +// recorded, and CI is left red as it was. The budget is per attempt: a +// repair round's new head gets only what is left. +// +// Re-runs are per workflow run (`rerun-failed-jobs`), not per job as the +// plan's KTD5 first said: re-running one job puts its run in progress, so +// GitHub refuses every sibling in the same run until it finishes, and N +// failed matrix shards would cost N workflow durations. +package worker + +import ( + "context" + "fmt" + "time" + + "github.com/StructuPath/jig/internal/protocol" +) + +// CIRerunSummary is one re-run as the publish summary reports it. The field +// names are pinned: the report reads ci_reruns[].{attempt, jobs, outcome}. +type CIRerunSummary struct { + // Attempt counts re-runs within the attempt, from 1. + Attempt int `json:"attempt"` + // Head is the commit the jobs were re-run on. + Head string `json:"head"` + // Jobs names the failed jobs re-run (or, when none were sent, the ones + // that would have been). + Jobs []string `json:"jobs"` + // Outcome is passed, failed, or the code the re-run stopped on. + Outcome string `json:"outcome"` + Detail string `json:"detail,omitempty"` +} + +const ( + ciRerunPassed = "passed" + ciRerunFailed = "failed" + // ciRerunRefused: GitHub refused a re-run for a reason other than a + // workflow run still in progress. + ciRerunRefused = "ci_rerun_refused" + // ciRerunHeadMoved: someone pushed to the branch while jig waited to + // re-run; there is nothing of jig's left to re-run. + ciRerunHeadMoved = "ci_rerun_head_moved" + // ciRerunInProgress is the gateway's code for GitHub refusing a re-run + // because the workflow run has not finished. It is waited out, never + // recorded as a spent re-run. + ciRerunInProgress = "ci_rerun_in_progress" + // ciRerunCancelled: the job was cancelled before a re-run request. + ciRerunCancelled = "ci_rerun_cancelled" + // maxCIRerunInProgressRetries bounds how often one run's re-run is asked + // for again after an "in progress" refusal; the CI timeout bounds how + // long. + maxCIRerunInProgressRetries = 5 +) + +// isActionsJob reports whether a check is a GitHub Actions job jig can +// re-run: only those have a job id, and it is their check-run id. +func isActionsJob(check CICheck) bool { + return check.App == "github-actions" && check.CheckRunID > 0 +} + +// allActionsJobs reports whether every red check is a re-runnable Actions +// job. One red check of any other kind means a re-run cannot turn CI green, +// so none is attempted (R7). +func allActionsJobs(checks []CICheck) bool { + if len(checks) == 0 { + return false + } + for _, check := range checks { + if check.Verdict == CIFail && !isActionsJob(check) { + return false + } + } + return true +} + +// rerunView is how the CI wait reads a head after re-runs: a check run jig +// just re-ran still shows its old red result until GitHub creates the new +// run, so it counts as pending until a check run of the same name that was +// not on the head before the re-run appears (KTD5). +type rerunView struct { + known map[int64]bool // every check-run id on the head before the re-run + rerun map[int64]bool // the check-run ids jig asked GitHub to re-run +} + +func newRerunView(checks []CICheck) *rerunView { + view := &rerunView{known: make(map[int64]bool, len(checks)), rerun: map[int64]bool{}} + for _, check := range checks { + if check.CheckRunID > 0 { + view.known[check.CheckRunID] = true + } + } + return view +} + +// apply rewrites one poll's checks. A re-run check is dropped once as many +// new runs of its name exist as were re-run under it, and is pending until +// then. A nil view changes nothing. +func (v *rerunView) apply(checks []CICheck) []CICheck { + if v == nil || len(v.rerun) == 0 { + return checks + } + fresh, stale := map[string]int{}, map[string]int{} + for _, check := range checks { + switch { + case v.rerun[check.CheckRunID]: + stale[check.Name]++ + case check.CheckRunID > 0 && !v.known[check.CheckRunID]: + fresh[check.Name]++ + } + } + viewed := make([]CICheck, 0, len(checks)) + for _, check := range checks { + if v.rerun[check.CheckRunID] { + if fresh[check.Name] >= stale[check.Name] { + continue + } + check.Verdict = CIPending + } + viewed = append(viewed, check) + } + return viewed +} + +// rerunFlakyCI runs declared re-runs while CI is red on the head jig pushed, +// every red check is an Actions job, and budget remains. summary is a +// publish summary that ended on ci_failed; the returned one is either +// published (flaky), still ci_failed (for a repair round or the end), or +// ended on the code a re-run stopped on (a moved head, a lost lease, a +// cancellation). Every pass through the loop records one ci_reruns entry or +// returns, so it runs at most the budget's times, and each pass is bounded +// by one CI timeout. +func (w *Worker) rerunFlakyCI( + ctx context.Context, + gateway PullRequestGateway, + options PublishOptions, + target publishTarget, + summary PublishSummary, + ci ciPolicy, +) PublishSummary { + for summary.Code == "ci_failed" && len(summary.CIReruns) < ci.rerunBudget { + head := summary.CIRef + if head == "" || head != summary.RemoteRef || !allActionsJobs(summary.CIFailures) { + // Red on a head jig did not push is a person's to fix; red on a + // check that is not an Actions job cannot be re-run. + return summary + } + entry := CIRerunSummary{Attempt: len(summary.CIReruns) + 1, Head: head, + Jobs: checkNames(summary.CIFailures)} + // One deadline for the whole re-run: settle, requests, judgement. + deadline := time.Now().Add(ci.timeout) + + // The CI wait stopped at the first red check; its siblings may still + // be running, and GitHub will not re-run a job in a running workflow. + checks, err := w.settleCI(ctx, gateway, options, target, head, deadline, nil) + if err != nil { + return rerunStopped(summary, entry, err) + } + failed := failedChecks(checks) + if len(failed) == 0 { + // Nothing is red any more (someone re-ran it). Judge the head + // afresh; with no re-run of jig's, a pass here is not flaky, and + // a judgement that cannot finish leaves CI red as it was. + judged := w.judgeAfterRerun(ctx, gateway, options, target, summary, head, deadline, nil) + if unjudged(judged.Code) { + return summary + } + return judged + } + summary.CIFailures = boundedChecks(failed) + entry.Jobs = checkNames(failed) + if !allActionsJobs(failed) { + // A sibling that finished red is not an Actions job. + return summary + } + + view := newRerunView(checks) + if err := w.requestReruns(ctx, gateway, options, target, head, deadline, summary.CIFailures, view); err != nil { + return rerunStopped(summary, entry, err) + } + judged := w.judgeAfterRerun(ctx, gateway, options, target, summary, head, deadline, view) + switch { + case judged.Published(): + judged.CIFlaky = true + entry.Outcome = ciRerunPassed + case judged.Code == "ci_failed": + entry.Outcome = ciRerunFailed + entry.Detail = judged.Detail + case unjudged(judged.Code): + // GitHub took the re-run but it never finished (an outage, a + // queued concurrency group) or CI could not be read: that is + // no verdict on the code, so CI stays red exactly as it was and + // a repair round, or the plain red stop, takes it from here. + entry.Outcome = judged.Code + entry.Detail = judged.Detail + summary.CIReruns = append(summary.CIReruns, entry) + return summary + default: + entry.Outcome = judged.Code + entry.Detail = judged.Detail + } + summary = judged + summary.CIReruns = append(summary.CIReruns, entry) + } + return summary +} + +// unjudged reports whether a re-run's judgement ended without a verdict on +// the code: CI did not finish in time, or could not be read. +func unjudged(code string) bool { + return code == "ci_timeout" || code == "ci_unavailable" +} + +// rerunStopped records a re-run that stopped before its judgement. A lost +// lease, a cancellation, and a moved head end the publish on that code; any +// other stop (a refusal, CI that never settled) leaves CI red, so a repair +// round or the attempt's end takes it from there. Either way the entry is +// recorded, and the loop does not go round again. +func rerunStopped(summary PublishSummary, entry CIRerunSummary, err error) PublishSummary { + code := publishCode(err) + entry.Outcome = code + entry.Detail = boundedText(err.Error(), protocol.MaxPublishDiagnosticBytes) + summary.CIReruns = append(summary.CIReruns, entry) + if code == ciRerunRefused || unjudged(code) { + return summary + } + return failSummary(summary, code, err.Error()) +} + +// requestReruns asks GitHub to re-run the failed jobs of each workflow run +// they belong to, once per run. Every request is fenced: the lease is +// freshened and checked for cancellation or loss immediately before it, so +// an attempt that lost its lease or was cancelled sends none. An "in +// progress" refusal is waited out and asked again, while the deadline +// allows and at most maxCIRerunInProgressRetries times; any other refusal +// stops the re-run. +func (w *Worker) requestReruns( + ctx context.Context, gateway PullRequestGateway, options PublishOptions, + target publishTarget, head string, deadline time.Time, failed []CICheck, view *rerunView, +) error { + runs, order := map[int64][]CICheck{}, []int64{} + for _, job := range failed { + run, err := gateway.ActionsJobRun(ctx, target.repository, job.CheckRunID) + if err != nil { + return publishFailure(ciRerunRefused, "the workflow run of %s could not be read: %s", job.Name, err.Error()) + } + if _, seen := runs[run]; !seen { + order = append(order, run) + } + runs[run] = append(runs[run], job) + } + for _, run := range order { + jobs := runs[run] + for retries := 0; ; retries++ { + if err := fenceRerun(ctx, target); err != nil { + return err + } + err := gateway.RerunFailedJobs(ctx, target.repository, run) + if err == nil { + for _, job := range jobs { + view.rerun[job.CheckRunID] = true + } + break + } + if publishCode(err) != ciRerunInProgress { + return publishFailure(ciRerunRefused, "GitHub refused to re-run %s: %s", + describeChecks(jobs), err.Error()) + } + if retries == maxCIRerunInProgressRetries { + return publishFailure(ciRerunRefused, + "GitHub still reported the workflow run of %s in progress after %d waits: %s", + describeChecks(jobs), retries, err.Error()) + } + if !time.Now().Before(deadline) { + return publishFailure(ciRerunRefused, + "GitHub still reported the workflow run of %s in progress at the CI timeout: %s", + describeChecks(jobs), err.Error()) + } + if err := w.ciPause(ctx, options, target, head, deadline); err != nil { + return err + } + if _, err := w.settleCI(ctx, gateway, options, target, head, deadline, view); err != nil { + return err + } + } + } + return nil +} + +// fenceRerun is the lease fence before one re-run request: heartbeat if the +// lease may be stale, then refuse if the lease is known lost or the job was +// cancelled — a heartbeat reports cancellation without failing, and a fresh +// lease is not heartbeated at all, so both are checked here. +func fenceRerun(ctx context.Context, target publishTarget) error { + if err := target.lease.freshen(ctx); err != nil { + return fmt.Errorf("lease could not be freshened: %w", err) + } + if err := target.lease.lostVerdict(); err != nil { + return fmt.Errorf("the lease was lost: %w", err) + } + select { + case <-target.lease.cancelled: + return publishFailure(ciRerunCancelled, "the job was cancelled before its failed jobs were re-run") + default: + return nil + } +} + +// judgeAfterRerun judges the head the re-run ran on — never whatever the +// branch has moved to — within the re-run's deadline, and records a green +// verdict through the fenced ci step. A branch that moves after the re-run +// ends it with ci_rerun_head_moved: a person's push is theirs to judge, not +// jig's flaky pass. +func (w *Worker) judgeAfterRerun( + ctx context.Context, gateway PullRequestGateway, options PublishOptions, + target publishTarget, summary PublishSummary, head string, deadline time.Time, view *rerunView, +) PublishSummary { + summary.CIRef = head + checks, err := w.settleCI(ctx, gateway, options, target, head, deadline, view) + if err == nil && len(checks) == 0 { + err = publishFailure("ci_unavailable", "no CI checks were reported on %s after the re-run", shortSHA(head)) + } + if err == nil { + if failed := failedChecks(checks); len(failed) > 0 { + summary.CIFailures = boundedChecks(failed) + err = publishFailure("ci_failed", "%d CI check(s) failed on %s after the re-run: %s", + len(failed), shortSHA(head), describeChecks(failed)) + } + } + if err != nil { + return failSummary(summary, publishCode(err), + fmt.Sprintf("publish step %q: %s", protocol.PublishStepCI, err.Error())) + } + summary.CIFailures = nil + green, err := w.publishStep(ctx, target, protocol.PublishStepCI, map[string]protocol.PublishRecord{}, &summary, + func(authorization protocol.PublishAuthorization) (protocol.PublishStepRequest, error) { + return protocol.PublishStepRequest{Step: protocol.PublishStepCI, + Branch: authorization.Branch, RemoteRef: head}, nil + }) + if err != nil { + return summary + } + summary.CIRef = green.RemoteRef + summary.State, summary.Code, summary.Detail = PublishStatePublished, "", "" + return summary +} + +// settleCI polls one head until none of its checks is pending, and returns +// them. Unlike the CI wait it does not stop at a red check, and it does not +// follow the branch: a head that moves ends it with ci_rerun_head_moved. +func (w *Worker) settleCI( + ctx context.Context, gateway PullRequestGateway, options PublishOptions, + target publishTarget, head string, deadline time.Time, view *rerunView, +) ([]CICheck, error) { + transient := 0 + for { + current, err := w.remoteHead(ctx, target, target.branch) + if err == nil && current != head { + return nil, publishFailure(ciRerunHeadMoved, + "the branch moved from %s to %s while jig waited to re-run its failed jobs", + shortSHA(head), shortSHA(current)) + } + var checks []CICheck + if err == nil { + checks, err = gateway.CommitChecks(ctx, target.repository, head) + } + var pending []CICheck + if err != nil { + transient++ + if transient >= protocol.MaxCITransientFailures { + return nil, publishFailure("ci_unavailable", + "CI state could not be read %d times in a row: %s", + transient, boundedText(err.Error(), protocol.MaxPublishDiagnosticBytes)) + } + } else { + transient = 0 + checks = view.apply(checks) + for _, check := range checks { + if check.Verdict == CIPending { + pending = append(pending, check) + } + } + if len(pending) == 0 { + return checks, nil + } + } + if !time.Now().Before(deadline) { + return nil, publishFailure("ci_timeout", + "CI on %s did not finish within the re-run's CI timeout; still pending: %s", + shortSHA(head), describeChecks(pending)) + } + if err := w.ciPause(ctx, options, target, head, deadline); err != nil { + return nil, err + } + } +} + +func failedChecks(checks []CICheck) []CICheck { + var failed []CICheck + for _, check := range checks { + if check.Verdict == CIFail { + failed = append(failed, check) + } + } + return failed +} diff --git a/internal/worker/publish_rerun_test.go b/internal/worker/publish_rerun_test.go new file mode 100644 index 0000000..7e9b816 --- /dev/null +++ b/internal/worker/publish_rerun_test.go @@ -0,0 +1,966 @@ +// publish_rerun_test.go — declared re-runs of flaky Actions checks (plan +// U4) through the real worker, control plane, and git remote, with CI and +// the re-run endpoint scripted by the fake gateway. +package worker + +import ( + "context" + "errors" + "fmt" + "os" + "path/filepath" + "slices" + "strings" + "testing" + "time" + + "github.com/StructuPath/jig/internal/protocol" +) + +// rerunSnapshot waits for CI with a re-run budget and, when repairBudget is +// positive, a repair loop too. +func rerunSnapshot(rerunBudget, repairBudget int) string { + if repairBudget == 0 { + return integrationSnapshot + fmt.Sprintf(`publish: + ci: + wait: true + rerun: {budget: %d} +`, rerunBudget) + } + return integrationSnapshot + fmt.Sprintf(` - name: review + kind: agent + owner: builder +publish: + ci: + wait: true + rerun: {budget: %d} + on_fail: {run: build, budget: %d} +`, rerunBudget, repairBudget) +} + +func actionsJob(name, verdict string, id int64) CICheck { + check := check(name, verdict) + check.App = "github-actions" + check.CheckRunID = id + return check +} + +// flakyCI scripts CI per head: script is called with the 0-based index of +// the head being judged (in the order heads are first seen), the number of +// re-run requests made so far, and the 1-based poll count. It runs under +// the gateway's lock, so it reads g.reruns directly. +type flakyCI struct { + heads []string + script func(head, reruns, poll int) []CICheck +} + +func (c *flakyCI) install(g *fakeGateway) { + g.checks = func(sha string, poll int) ([]CICheck, error) { + index := slices.Index(c.heads, sha) + if index < 0 { + c.heads = append(c.heads, sha) + index = len(c.heads) - 1 + } + return c.script(index, len(g.reruns), poll), nil + } +} + +func newRerunScenario(t *testing.T, rerunBudget, repairBudget int, script func(head, reruns, poll int) []CICheck, + rounds ...roundScript) (*repairScenario, *flakyCI) { + t.Helper() + s := &repairScenario{h: newHarness(t), ci: &ciScript{red: map[string]bool{}}} + var head, identity string + s.originDir, head, identity = newOriginRepo(t) + s.h.seedRunWithSnapshot("run-1", rerunSnapshot(rerunBudget, repairBudget), + protocol.RunTarget{Repository: identity, BaseSHA: head}) + s.job = s.h.enqueue("run-1", identity) + s.branch = protocol.PublishBranch(s.job.ID, 1) + s.continuation = &scriptedContinuation{t: t, rounds: rounds} + s.gateway = newFakeGateway() + ci := &flakyCI{script: script} + ci.install(s.gateway) + s.w = newCIWorker(t, s.h, repairRunner(t, s.continuation), s.gateway) + return s, ci +} + +// flakyTest is red until the first re-run, then its new run (a new check-run +// id) passes. lint is a green Actions sibling throughout. +func flakyTest(_, reruns, _ int) []CICheck { + if reruns == 0 { + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIFail, 11)} + } + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIPass, 12)} +} + +// brokenTest is red on every run, each re-run a new check-run id. +func brokenTest(head, reruns, _ int) []CICheck { + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIFail, int64(100*head+11+reruns))} +} + +// forbidReruns fails the test on any re-run request, and refuses it so the +// run under test ends promptly instead of waiting on a re-run that never +// lands. +func forbidReruns(t *testing.T) func(int64, int) error { + return func(jobID int64, _ int) error { + t.Errorf("a re-run of job %d was requested", jobID) + return publishFailure("gh_failed", "re-runs are forbidden in this test") + } +} + +func assertReruns(t *testing.T, summary PublishSummary, want ...CIRerunSummary) { + t.Helper() + if len(summary.CIReruns) != len(want) { + t.Fatalf("ci_reruns = %+v, want %d entries", summary.CIReruns, len(want)) + } + for i, entry := range want { + got := summary.CIReruns[i] + if got.Attempt != entry.Attempt || got.Outcome != entry.Outcome || + (entry.Head != "" && got.Head != entry.Head) || + (entry.Jobs != nil && !slices.Equal(got.Jobs, entry.Jobs)) || + (entry.Detail != "" && !strings.Contains(got.Detail, entry.Detail)) { + t.Fatalf("ci_reruns[%d] = %+v, want %+v", i, got, entry) + } + } +} + +// A single flaky Actions job that passes on re-run publishes green, flagged +// flaky, and no repair round runs even though one was declared (R5, R6). +func TestAFlakyActionsJobPassesOnRerunAndPublishes(t *testing.T) { + s, ci := newRerunScenario(t, 1, 1, flakyTest, fixRound()) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.Published() || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted, published, and flagged flaky", attempt.State, summary) + } + pushed := remoteBranches(t, s.originDir)[s.branch] + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Head: pushed, Jobs: []string{"test"}, Outcome: "passed"}) + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{11}) { + t.Fatalf("re-run requests = %v, want only the failed job 11", requests) + } + if len(s.continuation.failures) != 0 || len(s.rounds(t, attempt.ID)) != 0 || len(summary.CIRepairs) != 0 { + t.Fatal("a repair round ran although the re-run went green") + } + if len(ci.heads) != 1 || summary.CIRef != pushed || summary.RemoteRef != pushed { + t.Fatalf("heads judged = %v ci_ref = %s, want CI judged green on the one pushed head %s", + ci.heads, summary.CIRef, pushed) + } + if steps := recordedSteps(t, s.w, attempt.ID); !slices.Contains(steps, protocol.PublishStepCI) { + t.Fatalf("ledger steps = %v, want ci recorded after the re-run", steps) + } + if !strings.Contains(attempt.Result, `"ci_reruns":[{"attempt":1,`) || + !strings.Contains(attempt.Result, `"jobs":["test"],"outcome":"passed"`) { + t.Fatalf("result = %s, want the pinned ci_reruns shape", attempt.Result) + } +} + +// A sibling still running when the first check goes red delays the re-run +// until it finishes: GitHub refuses to re-run a job in a running workflow. +func TestARerunWaitsForAPendingSiblingToFinish(t *testing.T) { + const siblingDoneAt = 4 + var polls int + s, _ := newRerunScenario(t, 1, 0, func(_, reruns, poll int) []CICheck { + polls = poll + slow := actionsJob("slow-e2e", CIPending, 21) + if poll >= siblingDoneAt { + slow = actionsJob("slow-e2e", CIPass, 21) + } + if reruns == 0 { + return []CICheck{actionsJob("test", CIFail, 11), slow} + } + return []CICheck{actionsJob("test", CIPass, 12), slow} + }) + requestedAt := 0 + s.gateway.rerun = func(int64, int) error { + requestedAt = polls + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted after the re-run", attempt.State, summary) + } + if requestedAt < siblingDoneAt { + t.Fatalf("the re-run was requested at poll %d, before the sibling finished at poll %d", + requestedAt, siblingDoneAt) + } +} + +// The first polls after a re-run still return the old red check run: it is +// pending until a newer run of the same name appears, never read back red. +func TestTheStaleRedCheckAfterARerunIsPending(t *testing.T) { + var rerunAt int + s, _ := newRerunScenario(t, 1, 0, func(_, reruns, poll int) []CICheck { + lint := actionsJob("lint", CIPass, 1) + switch { + case reruns == 0: + return []CICheck{lint, actionsJob("test", CIFail, 11)} + case poll <= rerunAt+3: + // GitHub has not created the new run yet. + return []CICheck{lint, actionsJob("test", CIFail, 11)} + case poll <= rerunAt+5: + return []CICheck{lint, actionsJob("test", CIPending, 12)} + } + return []CICheck{lint, actionsJob("test", CIPass, 12)} + }) + s.gateway.rerun = func(int64, int) error { + rerunAt = s.gateway.checkPolls + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted { + t.Fatalf("attempt = %s summary = %+v, want the stale red run waited past", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "passed"}) + if requests := s.gateway.rerunRequests(); len(requests) != 1 { + t.Fatalf("re-run requests = %v, want one", requests) + } +} + +// A same-named check from another workflow that was already on the head is +// not the re-run's replacement: the re-run job stays pending until its own +// new run appears. +func TestAnOlderSameNamedCheckDoesNotReplaceTheRerun(t *testing.T) { + view := newRerunView([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIPass, 12)}) + view.rerun[11] = true + viewed := view.apply([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIPass, 12)}) + if len(viewed) != 2 || viewed[0].Verdict != CIPending { + t.Fatalf("viewed = %+v, want the re-run check pending beside the other workflow's", viewed) + } + viewed = view.apply([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIPass, 12), + actionsJob("build", CIPass, 13)}) + if len(viewed) != 2 || viewed[0].CheckRunID != 12 || viewed[1].CheckRunID != 13 { + t.Fatalf("viewed = %+v, want the stale run dropped once its new run exists", viewed) + } + // Two same-named jobs re-run (one per workflow): one new run is not + // both replacements, so both stay pending until the second appears. + view = newRerunView([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIFail, 12)}) + view.rerun[11], view.rerun[12] = true, true + viewed = view.apply([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIFail, 12), + actionsJob("build", CIPass, 13)}) + if len(viewed) != 3 || viewed[0].Verdict != CIPending || viewed[1].Verdict != CIPending { + t.Fatalf("viewed = %+v, want both re-run checks pending until both are replaced", viewed) + } + viewed = view.apply([]CICheck{actionsJob("build", CIFail, 11), actionsJob("build", CIFail, 12), + actionsJob("build", CIPass, 13), actionsJob("build", CIPass, 14)}) + if len(viewed) != 2 { + t.Fatalf("viewed = %+v, want both stale runs dropped once both new runs exist", viewed) + } + if plain := (*rerunView)(nil).apply([]CICheck{actionsJob("build", CIFail, 11)}); plain[0].Verdict != CIFail { + t.Fatal("a nil view rewrote a check") + } +} + +// GitHub refusing because the workflow run is still in progress is waited +// out and asked again, and does not spend the budget. +func TestAnInProgressRefusalIsRetriedWithoutSpendingTheBudget(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, flakyTest) + var pollsAt []int + s.gateway.rerun = func(_ int64, call int) error { + pollsAt = append(pollsAt, s.gateway.checkPolls) + if call <= 2 { + return publishFailure(ciRerunInProgress, "This workflow is already running") + } + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted after the retried re-run", attempt.State, summary) + } + for i := 1; i < len(pollsAt); i++ { + if pollsAt[i] <= pollsAt[i-1] { + t.Fatalf("CI polls at each request = %v: a refusal was asked again without waiting on CI", pollsAt) + } + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "passed"}) + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{11, 11, 11}) { + t.Fatalf("re-run requests = %v, want job 11 asked three times", requests) + } +} + +// An "in progress" refusal that never clears is bounded: after the retries +// run out it is recorded as a refusal. +func TestAnEndlessInProgressRefusalIsBounded(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, flakyTest) + s.gateway.rerun = func(int64, int) error { + return publishFailure(ciRerunInProgress, "This workflow is already running") + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "ci_rerun_refused", Detail: "in progress"}) + if requests := s.gateway.rerunRequests(); len(requests) != maxCIRerunInProgressRetries+1 { + t.Fatalf("re-run requests = %d, want %d", len(requests), maxCIRerunInProgressRetries+1) + } +} + +// Any other refusal is recorded with its diagnostic and falls through: to a +// repair round when one is declared, to ci_failed when not. +func TestAnyOtherRefusalIsRecordedAndFallsThrough(t *testing.T) { + refuse := func(int64, int) error { + return publishFailure("gh_failed", "gh api job rerun failed: Resource not accessible by integration") + } + t.Run("to a repair round", func(t *testing.T) { + s, _ := newRerunScenario(t, 2, 1, func(head, reruns, poll int) []CICheck { + if head == 0 { + return brokenTest(head, reruns, poll) + } + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIPass, 111)} + }, fixRound()) + s.gateway.rerun = refuse + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted by the repair round, not flaky", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "ci_rerun_refused", + Detail: "Resource not accessible"}) + if len(summary.CIRepairs) != 1 || len(s.gateway.rerunRequests()) != 1 { + t.Fatalf("repairs = %+v requests = %v, want one round after one refused request", + summary.CIRepairs, s.gateway.rerunRequests()) + } + }) + t.Run("to ci_failed", func(t *testing.T) { + s, _ := newRerunScenario(t, 2, 0, brokenTest) + s.gateway.rerun = refuse + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "ci_rerun_refused"}) + if requests := s.gateway.rerunRequests(); len(requests) != 1 { + t.Fatalf("re-run requests = %v, want one: a refusal stops re-running this head", requests) + } + }) +} + +// A check still red after the budget goes on to a repair round when one is +// declared, and ends ci_failed with every re-run recorded when not. +func TestACheckStillRedAfterTheBudgetFallsThrough(t *testing.T) { + t.Run("to a repair round", func(t *testing.T) { + s, _ := newRerunScenario(t, 1, 1, func(head, reruns, poll int) []CICheck { + if head == 0 { + return brokenTest(head, reruns, poll) + } + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIPass, 111)} + }, fixRound()) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted by the round, not flaky", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "failed"}) + if len(summary.CIRepairs) != 1 || summary.CIRepairs[0].Outcome != "pushed" { + t.Fatalf("ci_repairs = %+v, want one pushed round", summary.CIRepairs) + } + if len(s.continuation.failures) != 1 || s.continuation.failures[0].Checks[0].CheckRunID != 12 { + t.Fatalf("the round was handed %+v, want the re-run's own red run", s.continuation.failures) + } + }) + t.Run("to ci_failed", func(t *testing.T) { + s, _ := newRerunScenario(t, 2, 0, brokenTest) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "failed"}, + CIRerunSummary{Attempt: 2, Outcome: "failed"}) + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{11, 12}) { + t.Fatalf("re-run requests = %v, want each red run re-run once, within the budget", requests) + } + }) +} + +// Any red check that is not an Actions job skips re-runs entirely (R7). +func TestARedNonActionsCheckSkipsReruns(t *testing.T) { + for name, red := range map[string][]CICheck{ + "commit status": {check("legacy/ci", CIFail)}, + "mixed": {actionsJob("test", CIFail, 11), check("legacy/ci", CIFail)}, + "no job id": {actionsJob("test", CIFail, 0)}, + "other app": {{Name: "codecov", Verdict: CIFail, App: "codecov", CheckRunID: 31}}, + } { + t.Run(name, func(t *testing.T) { + s, _ := newRerunScenario(t, 3, 0, func(int, int, int) []CICheck { return red }) + s.gateway.rerun = forbidReruns(t) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + if len(summary.CIReruns) != 0 || len(s.gateway.rerunRequests()) != 0 { + t.Fatalf("ci_reruns = %+v requests = %v, want none", summary.CIReruns, s.gateway.rerunRequests()) + } + }) + } +} + +// A red sibling that is not an Actions job, found only once CI settles, +// also stops the re-run before any request. +func TestARedNonActionsSiblingFoundWhileSettlingSkipsTheRerun(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, func(_, _, poll int) []CICheck { + legacy := check("legacy/ci", CIPending) + if poll >= 3 { + legacy = check("legacy/ci", CIFail) + } + return []CICheck{actionsJob("test", CIFail, 11), legacy} + }) + s.gateway.rerun = forbidReruns(t) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" { + t.Fatalf("attempt = %s summary = %+v, want ci_failed", attempt.State, summary) + } + if len(s.gateway.rerunRequests()) != 0 || len(summary.CIFailures) != 2 { + t.Fatalf("requests = %v failures = %+v, want no request and both red checks named", + s.gateway.rerunRequests(), summary.CIFailures) + } +} + +// The budget is per attempt: a repair round's new red head gets what the +// earlier head left, and nothing once it is spent. +func TestTheRerunBudgetCarriesAcrossRepairRounds(t *testing.T) { + t.Run("what is left applies to the new head", func(t *testing.T) { + s, ci := newRerunScenario(t, 2, 1, func(head, reruns, poll int) []CICheck { + if head == 0 { + return brokenTest(head, reruns, poll) + } + if reruns <= 1 { + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIFail, 111)} + } + return []CICheck{actionsJob("lint", CIPass, 1), actionsJob("test", CIPass, 112)} + }, fixRound()) + s.gateway.rerun = func(_ int64, call int) error { + if call == 1 { + return publishFailure("gh_failed", "refused") + } + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted, flaky on the fix", attempt.State, summary) + } + assertReruns(t, summary, + CIRerunSummary{Attempt: 1, Head: ci.heads[0], Outcome: "ci_rerun_refused"}, + CIRerunSummary{Attempt: 2, Head: ci.heads[1], Outcome: "passed"}) + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{11, 111}) { + t.Fatalf("re-run requests = %v, want one per head", requests) + } + }) + t.Run("a spent budget does not", func(t *testing.T) { + s, _ := newRerunScenario(t, 1, 1, brokenTest, fixRound()) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_repair_exhausted" { + t.Fatalf("attempt = %s summary = %+v, want ci_repair_exhausted", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "failed"}) + if requests := s.gateway.rerunRequests(); len(requests) != 1 || len(summary.CIRepairs) != 1 { + t.Fatalf("requests = %v repairs = %+v, want one re-run and one round", requests, summary.CIRepairs) + } + }) +} + +// A lease lost before a re-run means no request is sent — before each +// request, not only the first. +func TestALostLeaseSendsNoRerun(t *testing.T) { + for _, tc := range []struct { + name string + loseFrom int // the request count after which the lease is lost + want []int64 + }{ + {"before the first", 0, nil}, + {"between two jobs", 1, []int64{11}}, + } { + t.Run(tc.name, func(t *testing.T) { + w, gateway, target, pushed := publishedWorker(t) + var lost bool + // The attempt is already terminal, so a heartbeat is refused: + // a stale local clock is all it takes to lose the lease. + target.lease.now = func() time.Time { + if lost { + return time.Now().Add(time.Hour) + } + return time.Now() + } + lost = tc.loseFrom == 0 + gateway.checks = func(string, int) ([]CICheck, error) { + return []CICheck{actionsJob("test", CIFail, 11), actionsJob("lint", CIFail, 12)}, nil + } + gateway.rerun = func(_ int64, call int) error { + if call >= tc.loseFrom { + lost = true + } + return nil + } + summary := PublishSummary{State: PublishStateFailed, Code: "ci_failed", RemoteRef: pushed, CIRef: pushed, + CIFailures: []CICheck{actionsJob("test", CIFail, 11)}} + summary = w.rerunFlakyCI(context.Background(), gateway, fastCIOptions(), target, summary, + ciPolicy{wait: true, timeout: time.Minute, rerunBudget: 1}) + if requests := gateway.rerunRequests(); !slices.Equal(requests, tc.want) { + t.Fatalf("re-run requests = %v, want %v", requests, tc.want) + } + if summary.Published() || summary.Code == "ci_failed" || len(summary.CIReruns) != 1 || + summary.CIReruns[0].Outcome != summary.Code { + t.Fatalf("summary = %+v, want the publish ended on the lease verdict, recorded", summary) + } + }) + } +} + +// A person pushing while jig waits to re-run ends the re-run: there is +// nothing of jig's left on the branch to re-run. +func TestAPersonsPushWhileSettlingEndsTheRerun(t *testing.T) { + s, _ := newRerunScenario(t, 1, 1, nil, fixRound()) + s.gateway.rerun = forbidReruns(t) + var personal string + s.gateway.checks = func(_ string, poll int) ([]CICheck, error) { + if poll == 2 { + personal = pushToBranch(t, s.originDir, s.branch) + } + slow := actionsJob("slow", CIPending, 21) + if poll >= 4 { + slow = actionsJob("slow", CIPass, 21) + } + return []CICheck{actionsJob("test", CIFail, 11), slow}, nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_rerun_head_moved" { + t.Fatalf("attempt = %s summary = %+v, want ci_rerun_head_moved", attempt.State, summary) + } + if len(s.gateway.rerunRequests()) != 0 || len(s.continuation.failures) != 0 || + remoteBranches(t, s.originDir)[s.branch] != personal { + t.Fatal("a re-run or a repair round ran on top of a person's push") + } +} + +// CI red on a head jig did not push (a person pushed before CI was judged) +// is theirs to fix: no re-run is requested on it. +func TestRedCIOnAPersonsHeadIsNotRerun(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, nil) + s.gateway.rerun = forbidReruns(t) + var personal string + s.gateway.checks = func(_ string, poll int) ([]CICheck, error) { + if poll == 1 { + personal = pushToBranch(t, s.originDir, s.branch) + return []CICheck{actionsJob("test", CIPending, 11)}, nil + } + return []CICheck{actionsJob("test", CIFail, 11)}, nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" || summary.CIRef != personal { + t.Fatalf("attempt = %s summary = %+v, want ci_failed on the person's head %s", attempt.State, summary, personal) + } + if len(s.gateway.rerunRequests()) != 0 || len(summary.CIReruns) != 0 { + t.Fatalf("requests = %v ci_reruns = %+v, want no re-run on a person's head", + s.gateway.rerunRequests(), summary.CIReruns) + } +} + +// A publish-only retry never re-runs: like repair, it judges CI and nothing +// more. +func TestThePublishRetryNeverReruns(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, brokenTest) + if attempt, _ := s.run(t); attempt.State != protocol.AttemptAcceptedUnpublished { + t.Fatalf("attempt = %s, want accepted_unpublished", attempt.State) + } + retried, err := s.w.RetryPublish(context.Background(), s.job.ID, s.gateway, fastCIOptions()) + if err != nil { + t.Fatalf("publish retry: %v", err) + } + if summary := publishSummaryOf(t, retried.Result); summary.Code != "ci_failed" || len(summary.CIReruns) != 0 || + len(s.gateway.rerunRequests()) != 1 { + t.Fatalf("retry summary = %+v requests = %v, want red judged with no new re-run", + summary, s.gateway.rerunRequests()) + } +} + +// ---- review fixes -------------------------------------------------------------- + +// A person pushing after jig's re-run request is theirs to judge: the +// judgement stays pinned to the re-run head and ends ci_rerun_head_moved, +// never recording the person's green head as jig's flaky pass. +func TestAPersonsPushAfterTheRerunIsNotAFlakyPass(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, func(head, reruns, _ int) []CICheck { + if head == 0 && reruns == 0 { + return []CICheck{actionsJob("test", CIFail, 11)} + } + if head == 0 { + // jig's head would pass too: only the moved branch may stop + // it from being recorded as a flaky pass. + return []CICheck{actionsJob("test", CIPass, 12)} + } + return []CICheck{actionsJob("test", CIPass, 21)} + }) + var personal string + s.gateway.rerun = func(int64, int) error { + personal = pushToBranch(t, s.originDir, s.branch) + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_rerun_head_moved" || + summary.CIFlaky || summary.Published() { + t.Fatalf("attempt = %s summary = %+v, want ci_rerun_head_moved and not flaky", attempt.State, summary) + } + if len(summary.CIReruns) != 1 || summary.CIReruns[0].Outcome != "ci_rerun_head_moved" || + summary.CIReruns[0].Head == personal || summary.CIRef == personal { + t.Fatalf("summary = %+v, want the re-run recorded on jig's head %s, not the person's", summary, personal) + } + if steps := recordedSteps(t, s.w, attempt.ID); slices.Contains(steps, protocol.PublishStepCI) { + t.Fatalf("ledger steps = %v: CI was recorded green on a head jig did not re-run", steps) + } +} + +// A re-run whose CI cannot be read afterwards is no verdict on the code: CI +// stays red as it was, and the declared repair round still runs. +func TestAnUnreadableJudgementAfterARerunStillReachesRepair(t *testing.T) { + s, _ := newRerunScenario(t, 1, 1, nil, fixRound()) + ci := &flakyCI{} + s.gateway.checks = func(sha string, _ int) ([]CICheck, error) { + index := slices.Index(ci.heads, sha) + if index < 0 { + ci.heads = append(ci.heads, sha) + index = len(ci.heads) - 1 + } + switch { + case index > 0: + return []CICheck{actionsJob("test", CIPass, 111)}, nil + case len(s.gateway.reruns) == 0: + return []CICheck{actionsJob("test", CIFail, 11)}, nil + } + return nil, errors.New("connection reset") + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted by the repair round", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Head: ci.heads[0], Outcome: "ci_unavailable"}) + if len(summary.CIRepairs) != 1 || len(s.continuation.failures) != 1 || + s.continuation.failures[0].Checks[0].CheckRunID != 11 { + t.Fatalf("repairs = %+v handed = %+v, want one round handed the original red check", + summary.CIRepairs, s.continuation.failures) + } +} + +// One re-run — settle, request, and judgement — fits in one CI timeout, and +// a re-run GitHub accepted but never started leaves CI red as it was. +func TestARerunThatNeverFinishesIsBoundedAndLeavesCIRed(t *testing.T) { + w, gateway, target, pushed := publishedWorker(t) + start := time.Now() + const timeout = 400 * time.Millisecond + gateway.checks = func(string, int) ([]CICheck, error) { + slow := actionsJob("slow", CIPending, 21) + if time.Since(start) > timeout*3/4 { + slow = actionsJob("slow", CIPass, 21) + } + // The re-run is accepted, but its new run never appears. + return []CICheck{actionsJob("test", CIFail, 11), slow}, nil + } + red := []CICheck{actionsJob("test", CIFail, 11)} + summary := PublishSummary{State: PublishStateFailed, Code: "ci_failed", Detail: "1 CI check(s) failed", + RemoteRef: pushed, CIRef: pushed, CIFailures: red} + summary = w.rerunFlakyCI(context.Background(), gateway, fastCIOptions(), target, summary, + ciPolicy{wait: true, timeout: timeout, rerunBudget: 3}) + elapsed := time.Since(start) + if elapsed > timeout+timeout/2 { + t.Fatalf("the re-run took %s, more than one CI timeout of %s", elapsed, timeout) + } + if summary.Code != "ci_failed" || summary.CIRef != pushed || len(summary.CIFailures) != 1 || + summary.CIFailures[0].CheckRunID != 11 || summary.Detail != "1 CI check(s) failed" { + t.Fatalf("summary = %+v, want CI left red exactly as it was", summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Head: pushed, Outcome: "ci_timeout"}) + if requests := gateway.rerunRequests(); !slices.Equal(requests, []int64{11}) { + t.Fatalf("re-run requests = %v, want one, and no second re-run of a run still going", requests) + } +} + +// GitHub reporting no checks at all after a re-run is no green verdict: CI +// stays red as it was. +func TestNoChecksAfterARerunIsNotGreen(t *testing.T) { + s, _ := newRerunScenario(t, 1, 0, func(_, reruns, _ int) []CICheck { + if reruns == 0 { + return []CICheck{actionsJob("test", CIFail, 11)} + } + return nil + }) + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAcceptedUnpublished || summary.Code != "ci_failed" || summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want ci_failed, not a flaky pass", attempt.State, summary) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "ci_unavailable"}) +} + +// Failed jobs of one workflow run are re-run with one request for the run, +// not one per job (whose siblings GitHub would refuse while it runs). +func TestFailedJobsOfOneWorkflowRunAreRerunTogether(t *testing.T) { + var lastRequestAt int + s, _ := newRerunScenario(t, 1, 0, func(_, reruns, poll int) []CICheck { + if reruns < 2 || poll <= lastRequestAt+2 { + // Before the re-runs, and for a while after: GitHub still + // lists the old red runs of every re-run job. + return []CICheck{actionsJob("test (1)", CIFail, 11), actionsJob("test (2)", CIFail, 12), + actionsJob("lint", CIFail, 31)} + } + if poll <= lastRequestAt+4 { + // The run's new jobs appear one at a time: test (2) still + // shows its old red run, which is pending, not red. + return []CICheck{actionsJob("test (1)", CIPass, 13), actionsJob("test (2)", CIFail, 12), + actionsJob("lint", CIPass, 32)} + } + return []CICheck{actionsJob("test (1)", CIPass, 13), actionsJob("test (2)", CIPass, 14), + actionsJob("lint", CIPass, 32)} + }) + s.gateway.runOf = func(job int64) int64 { + if job == 31 { + return 3000 + } + return 1000 + } + s.gateway.rerun = func(run int64, _ int) error { + if run != 1000 && run != 3000 { + t.Errorf("re-run of run %d, want a workflow run id", run) + } + lastRequestAt = s.gateway.checkPolls + return nil + } + attempt, summary := s.run(t) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted and flaky", attempt.State, summary) + } + if requests := s.gateway.rerunRequests(); !slices.Equal(requests, []int64{1000, 3000}) { + t.Fatalf("re-run requests = %v, want one per workflow run", requests) + } + assertReruns(t, summary, CIRerunSummary{Attempt: 1, Outcome: "passed", + Jobs: []string{"test (1)", "test (2)", "lint"}}) +} + +// A job cancelled, or a lease the control plane already declared lost, sends +// no re-run even while the lease is fresh and freshen does not heartbeat. +func TestACancelledJobOrALostVerdictSendsNoRerunOnAFreshLease(t *testing.T) { + for _, tc := range []struct { + name string + setup func(*attemptLease) + code string + }{ + {"cancelled", func(l *attemptLease) { l.cancelOnce.Do(func() { close(l.cancelled) }) }, ciRerunCancelled}, + {"lost", func(l *attemptLease) { l.lost = errors.New("lease_not_owner") }, "publish_failed"}, + } { + t.Run(tc.name, func(t *testing.T) { + w, gateway, target, pushed := publishedWorker(t) + gateway.checks = func(string, int) ([]CICheck, error) { + return []CICheck{actionsJob("test", CIFail, 11)}, nil + } + tc.setup(target.lease) + summary := PublishSummary{State: PublishStateFailed, Code: "ci_failed", RemoteRef: pushed, CIRef: pushed, + CIFailures: []CICheck{actionsJob("test", CIFail, 11)}} + summary = w.rerunFlakyCI(context.Background(), gateway, fastCIOptions(), target, summary, + ciPolicy{wait: true, timeout: time.Minute, rerunBudget: 1}) + if requests := gateway.rerunRequests(); len(requests) != 0 { + t.Fatalf("re-run requests = %v, want none", requests) + } + if summary.Code != tc.code || len(summary.CIReruns) != 1 || summary.CIReruns[0].Outcome != tc.code { + t.Fatalf("summary = %+v, want it ended on %s, recorded", summary, tc.code) + } + }) + } +} + +// A heartbeat that receives a lost-lease verdict records it, so the next +// re-run fence refuses even though the lease now looks fresh. +func TestAHeartbeatRecordsTheLostVerdict(t *testing.T) { + w, _, target, _ := publishedWorker(t) + // A well-formed token that is not the attempt's: the control plane's + // answer is a verdict, not a malformed request. + target.lease = newAttemptLease(w.client, target.attemptID, strings.Repeat("a", 64)) + if target.lease.lostVerdict() != nil { + t.Fatal("a new lease reports a lost verdict") + } + target.lease.now = func() time.Time { return time.Now().Add(time.Hour) } + if err := target.lease.freshen(context.Background()); err == nil || !leaseLost(err) { + t.Fatalf("freshen = %v, want a lost-lease verdict from the terminal attempt", err) + } + target.lease.now = time.Now + if target.lease.lostVerdict() == nil { + t.Fatal("the heartbeat's lost verdict was not recorded") + } + if err := fenceRerun(context.Background(), target); err == nil { + t.Fatal("the re-run fence passed a lease the control plane declared lost") + } +} + +// Past the deadline an "in progress" refusal is not asked again: no burst of +// back-to-back requests. +func TestNoInProgressRetryPastTheDeadline(t *testing.T) { + w, gateway, target, pushed := publishedWorker(t) + gateway.checks = func(string, int) ([]CICheck, error) { + return []CICheck{actionsJob("test", CIFail, 11)}, nil + } + gateway.rerun = func(int64, int) error { + return publishFailure(ciRerunInProgress, "This workflow is already running") + } + failed := []CICheck{actionsJob("test", CIFail, 11)} + err := w.requestReruns(context.Background(), gateway, fastCIOptions(), target, pushed, + time.Now().Add(-time.Second), failed, newRerunView(failed)) + if publishCode(err) != ciRerunRefused || !strings.Contains(err.Error(), "CI timeout") { + t.Fatalf("err = %v, want a refusal at the CI timeout", err) + } + if requests := gateway.rerunRequests(); len(requests) != 1 { + t.Fatalf("re-run requests = %v, want exactly one past the deadline", requests) + } +} + +// ---- the gh gateway ---------------------------------------------------------- + +// The real gateway finds a job's workflow run, then re-runs the run's failed +// jobs in one request, and tells an "in progress" refusal from the rest. +func TestGitHubGatewayRerunsAWorkflowRunsFailedJobs(t *testing.T) { + var calls [][]string + var stdout, stderr []byte + var runErr error + gateway := &GitHubCLIGateway{ + LookPath: func(string) (string, error) { return "/usr/bin/gh", nil }, + Run: func(_ context.Context, name string, arguments ...string) ([]byte, []byte, bool, bool, error) { + calls = append(calls, append([]string{name}, arguments...)) + return stdout, stderr, false, false, runErr + }, + } + stdout = []byte("987654\n") + run, err := gateway.ActionsJobRun(context.Background(), "github.com/acme/widgets", 4242) + if err != nil || run != 987654 { + t.Fatalf("run = %d err = %v, want 987654", run, err) + } + want := []string{"gh", "api", "-H", "Accept: application/vnd.github+json", + "repos/acme/widgets/actions/jobs/4242", "--jq", ".run_id"} + if len(calls) != 1 || !slices.Equal(calls[0], want) { + t.Fatalf("gh calls = %v, want %v", calls, want) + } + for _, malformed := range []string{"", "null", "0", "abc"} { + stdout = []byte(malformed) + if _, err := gateway.ActionsJobRun(context.Background(), "github.com/acme/widgets", 4242); publishCode(err) != "gh_malformed_output" { + t.Fatalf("run id %q: err = %v, want gh_malformed_output", malformed, err) + } + } + + stdout, calls = nil, nil + if err := gateway.RerunFailedJobs(context.Background(), "github.com/acme/widgets", 987654); err != nil { + t.Fatalf("rerun: %v", err) + } + want = []string{"gh", "api", "-X", "POST", "-H", "Accept: application/vnd.github+json", + "repos/acme/widgets/actions/runs/987654/rerun-failed-jobs"} + if len(calls) != 1 || !slices.Equal(calls[0], want) { + t.Fatalf("gh calls = %v, want %v", calls, want) + } + + for _, tc := range []struct { + name string + stdout, stderr string + code string + }{ + {"in progress on stderr", "", "gh: This workflow is already running (HTTP 403)", ciRerunInProgress}, + {"in progress in the body", `{"message":"Cannot rerun a workflow run that is in progress"}`, + "gh: HTTP 403", ciRerunInProgress}, + {"forbidden", `{"message":"Resource not accessible by integration"}`, + "gh: Resource not accessible by integration (HTTP 403)", "gh_failed"}, + {"unauthenticated", "", "gh: To get started with GitHub CLI, please run: gh auth login", "gh_unauthenticated"}, + } { + t.Run(tc.name, func(t *testing.T) { + stdout, stderr, runErr = []byte(tc.stdout), []byte(tc.stderr), errors.New("exit status 1") + err := gateway.RerunFailedJobs(context.Background(), "github.com/acme/widgets", 987654) + if publishCode(err) != tc.code { + t.Fatalf("err = %v (code %s), want %s", err, publishCode(err), tc.code) + } + }) + } + + stdout, stderr, runErr = nil, nil, nil + calls = nil + if err := gateway.RerunFailedJobs(context.Background(), "github.com/acme/widgets", 0); publishCode(err) != "ci_rerun_invalid_id" { + t.Fatalf("err = %v, want ci_rerun_invalid_id", err) + } + if _, err := gateway.ActionsJobRun(context.Background(), "github.com/acme/widgets", -1); publishCode(err) != "ci_rerun_invalid_id" { + t.Fatalf("err = %v, want ci_rerun_invalid_id", err) + } + if err := gateway.RerunFailedJobs(context.Background(), "gitlab.com/acme/widgets", 1); publishCode(err) != "publish_unsupported_remote" { + t.Fatalf("err = %v, want publish_unsupported_remote", err) + } + if len(calls) != 0 { + t.Fatalf("gh was called for an invalid request: %v", calls) + } +} + +// ---- the gated real-gh test -------------------------------------------------- + +// rerunGateWorkflow fails both shards of its flaky matrix job on a workflow +// run's first attempt and passes on any re-run; its slow sibling keeps the +// run in progress well after the shards go red, which is the case the fakes +// could not see: GitHub refuses to re-run a job while its workflow run is +// still going. Two shards in one run prove the re-run is one request per +// run: per job, the second shard would be refused while the first re-ran. +const rerunGateWorkflow = `name: jig-rerun-gate +on: + push: + branches: ['jig/**'] +jobs: + flaky: + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + shard: [1, 2] + steps: + - run: test "${{ github.run_attempt }}" != "1" + slow: + runs-on: ubuntu-latest + steps: + - run: sleep 90 +` + +// TestCIRerunGate re-runs a real workflow run's failed jobs through real gh +// after its slow sibling finishes. It is skipped unless JIG_RERUN_GATE names a scratch +// repository (owner/repo) whose other workflows, if any, pass on a jig/** +// branch. The attempt's change is the gate workflow itself, so gh needs the +// workflow scope; the test opens a pull request and leaves it for the +// operator to close. +func TestCIRerunGate(t *testing.T) { + project := strings.TrimSpace(os.Getenv("JIG_RERUN_GATE")) + if project == "" { + t.Skip("set JIG_RERUN_GATE=owner/repo to run the flaky re-run gate") + } + ctx := context.Background() + identity := "github.com/" + project + stdout, err := runGit(ctx, "", "ls-remote", "https://github.com/"+project+".git", "HEAD") + if err != nil { + t.Fatalf("read %s HEAD: %v", project, err) + } + baseSHA, _, found := strings.Cut(strings.TrimSpace(stdout), "\t") + if !found || len(baseSHA) < 40 { + t.Fatalf("unreadable HEAD for %s: %q", project, stdout) + } + + h := newHarness(t) + h.seedRunWithSnapshot("run-gate", integrationSnapshot+`publish: + ci: + wait: true + timeout: 20m + rerun: {budget: 1} +`, protocol.RunTarget{Repository: identity, BaseSHA: baseSHA}) + h.enqueue("run-gate", identity) + options := PublishOptions{CommitAuthorName: "jig-test", CommitAuthorEmail: "jig@test", + CIPollInterval: 10 * time.Second} + runner := NewPublishingRunner(writeAndDeclare(t, map[string]string{ + ".github/workflows/jig-rerun-gate.yml": rerunGateWorkflow, + }), NewGitHubCLIGateway(), options) + w := newTestWorker(t, h, filepath.Join(t.TempDir(), "worker"), 1, runner) + runner.Bind(w) + + attempt, err := w.ClaimOnce(ctx) + if err != nil || attempt == nil { + t.Fatalf("claim: attempt=%v err=%v", attempt, err) + } + summary := publishSummaryOf(t, attempt.Result) + t.Logf("rerun gate: state=%s pull request=%s summary=%+v", attempt.State, summary.PullRequestURL, summary) + if attempt.State != protocol.AttemptAccepted || !summary.CIFlaky { + t.Fatalf("attempt = %s summary = %+v, want accepted and flagged flaky", attempt.State, summary) + } + jobs := []string(nil) + if len(summary.CIReruns) == 1 { + jobs = slices.Sorted(slices.Values(summary.CIReruns[0].Jobs)) + } + if len(summary.CIReruns) != 1 || summary.CIReruns[0].Outcome != "passed" || + !slices.Equal(jobs, []string{"flaky (1)", "flaky (2)"}) { + t.Fatalf("ci_reruns = %+v, want both flaky shards re-run once and passed", summary.CIReruns) + } +} diff --git a/internal/worker/publish_test.go b/internal/worker/publish_test.go index 3586ad2..f7fbecb 100644 --- a/internal/worker/publish_test.go +++ b/internal/worker/publish_test.go @@ -39,6 +39,39 @@ type fakeGateway struct { checks func(sha string, poll int) ([]CICheck, error) checkPolls int checkedSHA []string + // runOf maps an Actions job to its workflow run; nil puts every job in a + // run of its own whose id is the job id. + runOf func(jobID int64) int64 + // rerun scripts re-run requests: it is called with the workflow run id + // and the 1-based request count. Nil accepts every request. reruns + // lists the run ids of every request made, accepted or refused. + rerun func(runID int64, call int) error + reruns []int64 +} + +func (g *fakeGateway) ActionsJobRun(_ context.Context, _ string, jobID int64) (int64, error) { + g.mutex.Lock() + defer g.mutex.Unlock() + if g.runOf == nil { + return jobID, nil + } + return g.runOf(jobID), nil +} + +func (g *fakeGateway) RerunFailedJobs(_ context.Context, _ string, runID int64) error { + g.mutex.Lock() + defer g.mutex.Unlock() + g.reruns = append(g.reruns, runID) + if g.rerun == nil { + return nil + } + return g.rerun(runID, len(g.reruns)) +} + +func (g *fakeGateway) rerunRequests() []int64 { + g.mutex.Lock() + defer g.mutex.Unlock() + return append([]int64(nil), g.reruns...) } func newFakeGateway() *fakeGateway {