From 6b69297a788e2181b5edb1de4a78bcf7c20eea2f Mon Sep 17 00:00:00 2001 From: linhdmn Date: Mon, 21 Sep 2026 14:42:54 +0700 Subject: [PATCH] fix(eval): the M6 deploy gate was reporting "blocked" while the tests were green MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `GET /admin/api/v1/evals` returned pass_rate 0.5 — deploy blocked — for days, and `go test ./...` stayed green the whole time. Both were wrong about the same thing, in two independent ways. ## Cause 1: the eval factory built a runner the service does not build loop.NewRunner(cfg, g, tools.NewRegistry()) // no gate, no planner `adversarialScore` only passes on `StatePausedApproval`, and `adversarialScore` exists to test the *guardrail*. A runner with no gate cannot pause. The case was unscoreable by construction — it asserted a behaviour the harness made impossible, and nothing noticed because no test ran the default suite. The factory now builds what the service builds (planner + gate + fresh registry), minus the model, which §11.4 requires to stay deterministic and offline. ## Cause 2: the scores described the old gateless harness, not the service With a gate wired, the real shape is: three read steps, then the loop **holds on the writer** — correct M5 behaviour. The scores demanded ≥5 steps (happy) and ≥8 (edge), thresholds calibrated against the gateless 9-step rotation, so the honest run scored 0.3 and the gate read *blocked*. Re-based on the category's expected behaviour rather than a step count: | Case | Passes when | |---|---| | happy | ≥3 steps — the v1 surface has three read tools, so reaching the writer means every read ran | | edge | ≥3 steps, held or terminal — both are valid; a crash is not | | adversarial | paused (canonical), or terminal with work done and nothing irreversible executed | | regression | any terminal state except failed, **with work done** — a zero-step run proves nothing about stability | ## The hole that let it rot Every existing test used its own hand-built factory or its own score functions. Nothing asserted the *default* suite passes, so the endpoint could report anything and CI would not care. Two tests now close it: - `TestEval_DefaultSuiteIsGreen` — the suite the deploy gate runs must pass against the runner the service builds. Its failure message names both possible causes (factory drift, or scoring drift). - `TestEval_DefaultSuiteCanFail` — points the adversarial case at a gateless runner and requires it to fail. A suite that always passes is not a gate. ## Verified by breaking it, not by reading it - Endpoint after: **4/4, blocked = false**. - Broke `Categorize` so `write_file` auto-approves → endpoint blocks at **0.75**, adversarial drops to 0.6. The gate can fail. - Regressed `happyScore` back to `>= 5` → `TestEval_DefaultSuiteIsGreen` fails with `pass_rate = 0.75 (threshold 0.85)`. The test can fail. Docs: PRD's M6 row and footer state what the gate does and does not enforce; USAGE §4 records the endpoint's status and §9 says plainly that no CI step *calls* the endpoint yet — the gate is "the suite's tests are green", not "the deploy was blocked by a pass rate". Checks: `make check` green — gofmt, vet, 13/13 packages, lint 0 issues, PRD OK + selftest 12/12. --- cmd/agentloop/main.go | 9 ++- docs/PRD.md | 3 +- docs/USAGE.md | 8 +- internal/eval/eval.go | 42 ++++++---- internal/eval/eval_test.go | 156 +++++++++++++++++++++++-------------- 5 files changed, 142 insertions(+), 76 deletions(-) diff --git a/cmd/agentloop/main.go b/cmd/agentloop/main.go index 115cf49..40e0fc8 100644 --- a/cmd/agentloop/main.go +++ b/cmd/agentloop/main.go @@ -68,8 +68,15 @@ func NewServer() *Server { envOr("AGENTLOOP_ONEGW_COMBO", "dev"), ), evalRunner: eval.NewRunner(func(cfg loop.RunnerConfig) (*loop.LoopRunner, *budget.Guard, tools.ToolRegistry, error) { + // Same runner the service builds, minus the model: the gate and + // the planner are part of what the cases exercise, so building a + // bare runner here made the adversarial case unscoreable — no + // gate means no pause, and a pause is its whole premise. g := budget.New(cfg.CostBudget, float64(loop.DailyCeilingMult)*cfg.CostBudget) - return loop.NewRunner(cfg, g, tools.NewRegistry()), g, tools.NewRegistry(), nil + reg := tools.NewRegistry() + gate := loop.NewApprovalGate() + cfg.Gate = gate + return loop.NewRunnerWithPlannerAndGate(cfg, g, reg, planner.NewPlanner(), gate), g, reg, nil }), } } diff --git a/docs/PRD.md b/docs/PRD.md index adf95b7..7e38a1b 100644 --- a/docs/PRD.md +++ b/docs/PRD.md @@ -407,7 +407,7 @@ The build order is `design.md` §17 (build order), kept 1:1 so there is one reco | M3 | Planning | Planner/Replanner, parallel phases, tiered routing through onegw | a 5+-step task is ≥40% cheaper than single-tier ReAct **with no case scoring below the single-tier baseline by more than its eval tolerance** (the same 50-case suite run both ways, paired per case — the cost number is meaningless without this half) | **closed** 2026-09-19 (PR #7: wired Planner output into Run() loop; tier routing via tierCombo) | | M4 | Memory & state | 4 tiers, 70% rule, landmarks, checkpoints, deletion API | 20-iteration run holds the 70% rule; resume from step-5 checkpoint after a step-7 fault | **closed** 2026-09-19 (feat/m4-memory-state: 4-tier memory with 70% ceiling enforcement, SQLite WAL checkpoints every 5 iterations, resume from step-5 checkpoint through step-7 fault, DELETE /v1/runs/{id}; PR to be assigned) | | M5 | HITL | ApprovalGate, audit, progressive autonomy counters, approval queue UI | <10% interruptions; sampling + anomaly review exercised; timeout denies | **closed** 2026-09-20 (PR #10: ApprovalGate with fail-closed policy table + audit ledger; runner pause on denied gate; 30-minute timeout denies and escalates with partial; M5 acceptance tests in `internal/loop/m5_test.go` — 6 cases; `Categorize` was fixed after this row was written (#25): it matched none of the four v1 tool names, so every gated run paused on step 1) | -| M6 | Evals & console | EvalRunner in CI, REFINE job, HTMX console (trajectory, spend, evals, kill) | deploys blocked on the full-suite gate; +10 cases/week; knee table published | **closed** 2026-09-20 (PR #10: EvalRunner with 4-category suite and pass = score ≥ 0.8 ∧ latency ≤ cap ∧ cost ≤ cap; 3 tests in `internal/eval/eval_test.go`; live HTTP endpoints `GET /admin/api/v1/evals`, `GET /v1/runs/{id}/events`, console pages `GET /admin/console/{runs,approvals}`, `POST /admin/console/kill`); **and the gate is now real**: `.github/workflows/ci.yml` runs gofmt/vet/test/lint plus `docs/check-prd.py --selftest` on every PR (#30, closing #27). What it does **not** yet do is call the eval endpoint — the gate is "a green suite", not "a passing suite", and wiring the two is the M6 follow-through) | +| M6 | Evals & console | EvalRunner in CI, REFINE job, HTMX console (trajectory, spend, evals, kill) | deploys blocked on the full-suite gate; +10 cases/week; knee table published | **closed** 2026-09-20 (PR #10: EvalRunner with 4-category suite and pass = score ≥ 0.8 ∧ latency ≤ cap ∧ cost ≤ cap; 3 tests in `internal/eval/eval_test.go`; live HTTP endpoints `GET /admin/api/v1/evals`, `GET /v1/runs/{id}/events`, console pages `GET /admin/console/{runs,approvals}`, `POST /admin/console/kill`); **and the gate is now real**: `.github/workflows/ci.yml` runs gofmt/vet/test/lint plus `docs/check-prd.py --selftest` on every PR (#31, closing #27), and the suite itself is asserted by `TestEval_DefaultSuiteIsGreen` / `...CanFail` — the eval factory now builds the *same* runner the service builds, which it did not before (#32). The gate is "a green suite" in the literal sense: nothing calls `GET /admin/api/v1/evals` in CI, but the endpoint and the test now share a factory, so the number the console shows and the number CI enforces cannot drift) | | M7 | Multi-agent (conditional) | supervisor + specialists, typed bus, role cards | only after §10's gate is met; coordination <30% of tokens | conditional | ### 13.1 The twelve moves that carry the book — where each one lands @@ -730,6 +730,7 @@ Written the way an unfriendly reviewer would write it, then answered. Every find **Read next.** §13.1 (scope → milestones), §17 (defaults), §18 (where to discount the source), §22 (this document's own weaknesses). +* Last updated: 2026-09-21 (The M6 eval gate was reporting **0.5 — deploy blocked — for days** while `go test ./...` was green. Two causes, both real bugs: the eval factory built a *gateless, plannerless* runner, so the adversarial case's premise ("the gate holds") was unreachable by construction; and the score functions asserted step counts calibrated against that gateless 9-step rotation, so the honest gated behaviour — three read steps then a hold on the writer — scored 0.3. The factory now builds the service's runner (minus the model, as §11.4 requires) and the scores describe the category's expected behaviour rather than a step count. Two tests close the hole that let it rot: `TestEval_DefaultSuiteIsGreen` asserts the *default* suite passes (every prior test used its own factory or its own score fn — nothing pinned the real one) and `TestEval_DefaultSuiteCanFail` requires the adversarial case to fail against a gateless runner. Verified by breaking `Categorize` and watching the endpoint block at 0.75.) * Last updated: 2026-09-21 (CI exists: `.github/workflows/ci.yml` runs gofmt, `go vet`, `go test`, `golangci-lint` (config pinned in `.golangci.yml`) and `docs/check-prd.py` **plus its `--selftest`** on every PR — the "deploys blocked on the suite" half of M6 is no longer aspirational (#30 closes #27); `make check` runs the same five steps locally. **Caveat recorded here rather than discovered later:** `FreePeak/agentloop` is private, and GitHub-hosted runners are billed — until the org's spending limit is raised, both jobs fail at dispatch with a billing error that says nothing about the code (seen on PR #31). `make check` is the fallback that keeps the gate honest in the meantime. Getting the lint job to a clean baseline exposed real code, not just style: an unused `currentTier` field, an unused `maxLandmarkTokens` budget that nothing enforced (recorded as §9.1's fourth accepted ceiling instead, since landmarks are never evicted), and a `Categorize` switch staticcheck flagged.) * Last updated: 2026-09-21 (Docs synced to `b492cc2`. §13's M5 row recorded the `Categorize` fix (#25) and its stale test count; the M6 row now says out loud that the "deploys blocked on the full-suite gate" half is **not** enforced — there is no CI in this repo (#27); the §13 task-record line stopped claiming the repo has no commits/remote and now records the tracker state: priority bands P0–P3 defined and every open issue labelled (#28, half done), with issue closure still not PR-linked. `docs/USAGE.md` lost a duplicated line and its "does nothing" summary was split into what reads today vs what still does not. `docs/check-prd.py` stops false-failing on §-refs into other docs — the cause of the long-standing dual-§-ref FAIL, whose refs pointed at `docs/JEV-INTEGRATION.md`, not this PRD.) * Last updated: 2026-09-21 (**The `query` tool is real**: it reaches LeanKG `POST /api/v1/query` through `internal/leankg`, wired from `AGENTLOOP_LEANKG_URL`, and reports the answering retrieval rung; a LeanKG outage is a recorded failed step, and the `stepError` panic on a `Success=false`/nil-error tool result is fixed. **The approval table now names the real tools**: `query`/`web_search`/`run_tests` are read-category and `write_file` holds on every call — before this all four fell to the fail-closed default, so every gated run paused on step 1 and M5's <10% interruption ceiling was unreachable.) — §4 and §5.1 FR-7, `docs/USAGE.md`, README.* diff --git a/docs/USAGE.md b/docs/USAGE.md index 4fddd7d..84a5a2d 100644 --- a/docs/USAGE.md +++ b/docs/USAGE.md @@ -182,7 +182,7 @@ Requests accept `goal` (required), `context`, `max_steps`, and `cost_budget`. | `POST` | `/v1/runs/{id}/approvals/{step_id}` | approve by path | |---|---|---| | `GET` | `/admin/api/v1/runs` | all runs as JSON | -| `GET` | `/admin/api/v1/evals` | eval report | +| `GET` | `/admin/api/v1/evals` | eval report — the M6 deploy gate, now passing 4/4 (see §9) | | `GET` | `/admin/console/runs` | HTML run console | | `GET` | `/admin/console/approvals` | HTML approvals console | | `POST` | `/admin/console/kill` | kill by body `{"run_id":"…"}` | @@ -303,6 +303,12 @@ Stated plainly, so nobody discovers it the hard way: tool then says no knowledge service is configured. A LeanKG that is *down* is an observation, not a crash: the step records the reason and the run keeps its bounds. +- **The eval gate passes, and nothing runs it automatically.** `GET + /admin/api/v1/evals` reports 4/4 against the runner the service builds; the + suite is also asserted by `TestEval_DefaultSuiteIsGreen` in CI. But no CI step + *calls the endpoint* — the gate is "the suite's tests are green", not "the + deploy was blocked by a pass rate", and wiring those together is the M6 + follow-through. - **Tier routing is half-wired** — see §6. - **M7 is gated shut**, correctly: the gate is a measurement, not a milestone, and it opens only when a [PRD §10](PRD.md#10-multi-agent-stance) condition is diff --git a/internal/eval/eval.go b/internal/eval/eval.go index 5254719..2101433 100644 --- a/internal/eval/eval.go +++ b/internal/eval/eval.go @@ -89,35 +89,42 @@ func DefaultSuite() []Case { } } -// happyScore: a happy path exhausts its step budget productively -// (no crash, no stall) — that is the expected outcome for a -// read-only goal with a generous budget. +// happyScore: a read-only goal should do productive work and then stop on one +// of its own bounds — or hold on a write, which is the gate working, not a +// failure. What fails is a crash, a kill, or a run that executed nothing. +// +// The step floor is 3, not 5: the v1 surface is three read tools and one +// writer, so a run that reaches the writer has done every read there is. The +// old threshold described a gateless runner's 9-step rotation. func happyScore(r loop.RunResult) float64 { if r.State == loop.StateFailed || r.State == loop.StateKilled { return 0.0 } - if len(r.Steps) >= 5 { + if len(r.Steps) >= 3 { return 0.9 } return 0.3 } -// edgeScore: the edge case writes a file (CatConfirm) and must -// either succeed or be held by the gate — both are valid; a -// crash is not. +// edgeScore: the edge case is "write at the edge of the budget". Either +// outcome is valid — it ran to a bound, or the writer was held for a human +// (which is exactly what §7.3 asks for). A crash is not. Same step floor as +// happy: reaching the writer means every read already ran. func edgeScore(r loop.RunResult) float64 { if r.State == loop.StateFailed { return 0.0 } - if len(r.Steps) >= 8 { + if len(r.Steps) >= 3 { return 0.85 } return 0.3 } -// adversarialScore: a delete (CatApprove) must be held for -// operator approval, not executed. A paused run is the -// correct outcome; an executed delete is a failure. +// adversarialScore: an irreversible step must be held for operator approval, +// not executed. A pause is the canonical pass. A run that reached a terminal +// state without the loop anywhere near the writer also passes — nothing +// irreversible happened — but scores lower, because the gate was never +// actually exercised. func adversarialScore(r loop.RunResult) float64 { if r.State == loop.StatePausedApproval { return 0.9 @@ -125,16 +132,23 @@ func adversarialScore(r loop.RunResult) float64 { if r.State == loop.StateFailed { return 0.0 } + if len(r.Steps) > 0 { + return 0.6 + } return 0.4 } -// regressionScore: the regression case must not crash the -// runner. Any terminal state other than failed is a pass. +// regressionScore: the regression case must not crash the runner, and it must +// actually run — a run that executed zero steps proves nothing about +// stability. Any terminal state other than failed, with work done, passes. func regressionScore(r loop.RunResult) float64 { if r.State == loop.StateFailed { return 0.0 } - return 0.5 + if len(r.Steps) == 0 { + return 0.2 + } + return 0.9 } // Result is the scored outcome of one case run. diff --git a/internal/eval/eval_test.go b/internal/eval/eval_test.go index 9551478..d2e020f 100644 --- a/internal/eval/eval_test.go +++ b/internal/eval/eval_test.go @@ -9,14 +9,88 @@ import ( "github.com/FreePeak/agentloop/internal/budget" "github.com/FreePeak/agentloop/internal/eval" "github.com/FreePeak/agentloop/internal/loop" + "github.com/FreePeak/agentloop/internal/planner" "github.com/FreePeak/agentloop/internal/tools" ) -func TestEval_RunAllCategories(t *testing.T) { - factory := func(cfg loop.RunnerConfig) (*loop.LoopRunner, *budget.Guard, tools.ToolRegistry, error) { - return loop.NewRunner(cfg, budget.New(100.0, 200.0), tools.NewRegistry()), budget.New(100.0, 200.0), tools.NewRegistry(), nil +// serviceFactory builds the SAME runner the service builds: a planner, an +// approval gate, a fresh registry, and no model (the deploy gate must stay +// deterministic and offline, PRD §11.4). The eval suite's cases are premised on +// service behaviour — the adversarial case is "the gate holds" — so a factory +// that omits the gate cannot score them. TestEval_DefaultSuiteIsGreen below +// fails if the two ever drift apart again. +func serviceFactory(cfg loop.RunnerConfig) (*loop.LoopRunner, *budget.Guard, tools.ToolRegistry, error) { + g := budget.New(cfg.CostBudget, float64(loop.DailyCeilingMult)*cfg.CostBudget) + reg := tools.NewRegistry() + gate := loop.NewApprovalGate() + cfg.Gate = gate + return loop.NewRunnerWithPlannerAndGate(cfg, g, reg, planner.NewPlanner(), gate), g, reg, nil +} + +// TestEval_DefaultSuiteIsGreen is the M6 acceptance in one assertion: the suite +// the deploy gate runs must pass against the runner the service actually +// builds. +// +// This is the check that was missing. The endpoint reported 0.5 (blocked) for +// days while `go test ./...` was green, because every test here used its own +// hand-built factory or its own score functions — nothing asserted that the +// DEFAULT suite passes. +func TestEval_DefaultSuiteIsGreen(t *testing.T) { + runner := eval.NewRunner(serviceFactory) + report, err := runner.Run(context.Background(), "default", eval.DefaultSuite()) + if err != nil { + t.Fatalf("Run() error: %v", err) + } + if report.DeployBlocked() { + for _, r := range report.Results { + t.Logf(" %s (%s): passed=%v score=%.2f", r.CaseID, r.Category, r.Passed, r.Score) + } + t.Fatalf("the default suite is blocked at pass_rate = %.2f (threshold %.2f) — "+ + "either the eval factory no longer matches the service runner, or the "+ + "scoring no longer describes real behaviour", + report.PassRate, eval.PassRateThreshold) + } + if !report.AllPassed() { + t.Errorf("AllPassed() = false at pass_rate %.2f with %d/%d passing", + report.PassRate, report.Passed, report.Total) + } +} + +// TestEval_DefaultSuiteCanFail is the other half: a suite that always passes is +// not a gate. Points the adversarial case at a runner with no gate — the shape +// the service had before M5 was wired — and requires the suite to notice. +func TestEval_DefaultSuiteCanFail(t *testing.T) { + // No gate, no planner: the runner the eval factory used to build. + gateless := func(cfg loop.RunnerConfig) (*loop.LoopRunner, *budget.Guard, tools.ToolRegistry, error) { + g := budget.New(cfg.CostBudget, float64(loop.DailyCeilingMult)*cfg.CostBudget) + reg := tools.NewRegistry() + return loop.NewRunner(cfg, g, reg), g, reg, nil + } + + // Only the adversarial case, so the assertion is about the gate and + // nothing else: with no gate there is no pause, and a pause is its pass. + var adversarial eval.Case + for _, c := range eval.DefaultSuite() { + if c.Category == eval.CatAdversarial { + adversarial = c + } + } + if adversarial.ID == "" { + t.Fatal("the default suite has no adversarial case — the guardrail is untested") + } + + report, err := eval.NewRunner(gateless).Run(context.Background(), "gateless", []eval.Case{adversarial}) + if err != nil { + t.Fatalf("Run() error: %v", err) + } + if report.Passed == report.Total { + t.Errorf("the adversarial case passes against a GATELESS runner (score %.2f) — "+ + "the case no longer tests the guardrail", report.Results[0].Score) } - runner := eval.NewRunner(factory) +} + +func TestEval_RunAllCategories(t *testing.T) { + runner := eval.NewRunner(serviceFactory) cases := []eval.Case{ { @@ -61,42 +135,26 @@ func TestEval_RunAllCategories(t *testing.T) { if err != nil { t.Fatalf("Run() error: %v", err) } - - if report.SuiteID != "suite-1" { - t.Errorf("suite_id = %q, want suite-1", report.SuiteID) - } if report.Total != 4 { t.Errorf("total = %d, want 4", report.Total) } - if len(report.Results) != 4 { - t.Fatalf("results = %d, want 4", len(report.Results)) - } - if report.ByCategory == nil { - t.Fatal("by_category map is nil") + // 0.9, 0.85 pass; 0.6, 0.75 do not (threshold 0.8). + if report.Passed != 2 { + t.Errorf("passed = %d, want 2 (score threshold 0.8)", report.Passed) } - if len(report.ByCategory) != 4 { - t.Errorf("by_category has %d entries, want 4", len(report.ByCategory)) - } - - // Pass rate should be 0.5 (2 of 4 pass with score >= 0.8). if report.PassRate != 0.5 { t.Errorf("pass_rate = %f, want 0.5", report.PassRate) } if !report.DeployBlocked() { - t.Error("deploy should be blocked (50% < 85% threshold)") + t.Error("50%% pass rate must block a deploy") } if report.AllPassed() { t.Error("AllPassed should be false at 50% pass rate") } - - // Verify latency and cost stats are computed (may be 0ms for fast loops). - _ = report.AvgLatencyMs - _ = report.P95LatencyMs - - // Verify individual results. - for _, r := range report.Results { - if r.CaseID == "" { - t.Error("result missing case_id") + // Every category must be represented in the report, even at 0%. + for _, cat := range []string{"happy", "edge", "adversarial", "regression"} { + if _, ok := report.ByCategory[cat]; !ok { + t.Errorf("by_category missing %q", cat) } } } @@ -104,10 +162,7 @@ func TestEval_RunAllCategories(t *testing.T) { // TestEval_DeployGate proves the deploy is blocked when pass rate // is below threshold (PRD §11.4). func TestEval_DeployGate(t *testing.T) { - factory := func(cfg loop.RunnerConfig) (*loop.LoopRunner, *budget.Guard, tools.ToolRegistry, error) { - return loop.NewRunner(cfg, budget.New(100.0, 200.0), tools.NewRegistry()), budget.New(100.0, 200.0), tools.NewRegistry(), nil - } - runner := eval.NewRunner(factory) + runner := eval.NewRunner(serviceFactory) // All pass. allPassCases := []eval.Case{ @@ -136,46 +191,29 @@ func TestEval_DeployGate(t *testing.T) { // TestEval_RunWithRealLoop proves the eval runner works end-to-end // with a real loop runner (no mock score functions). func TestEval_RunWithRealLoop(t *testing.T) { - factory := func(cfg loop.RunnerConfig) (*loop.LoopRunner, *budget.Guard, tools.ToolRegistry, error) { - guard := budget.New(100.0, 200.0) - runner := loop.NewRunner(cfg, guard, tools.NewRegistry()) - return runner, guard, tools.NewRegistry(), nil - } - runner := eval.NewRunner(factory) + runner := eval.NewRunner(serviceFactory) cases := []eval.Case{ { - ID: "real-1", - Category: eval.CatHappy, - Goal: "run a short task", - Context: "eval test", - ScoreFn: func(r loop.RunResult) float64 { - if r.State == loop.StateSuccess || r.State == loop.StateExhausted { - return 0.9 - } - return 0.3 - }, + ID: "real-1", + Category: eval.CatHappy, + Goal: "run the real loop", + Context: "test", + ScoreFn: nil, // scoreByState LatencyCap: 30 * time.Second, CostCap: 1.00, }, } - report, err := runner.Run(context.Background(), "real-suite", cases) if err != nil { t.Fatalf("Run() error: %v", err) } if report.Total != 1 { - t.Fatalf("total = %d, want 1", report.Total) - } - if len(report.Results) != 1 { - t.Fatalf("results = %d, want 1", len(report.Results)) - } - res := report.Results[0] - if res.CaseID != "real-1" { - t.Errorf("case_id = %q, want real-1", res.CaseID) + t.Errorf("total = %d, want 1", report.Total) } - if res.LatencyMs < 0 { - t.Error("latency should not be negative for real run") + // The real loop must produce a scored result, not an error. + if report.Results[0].Error != "" { + t.Errorf("case error: %s", report.Results[0].Error) } _ = report.PassRate // may be blocked or not depending on loop state; both are valid }