From bec9ecdc4b429b8fb94513e1d311b7d08a44f6e9 Mon Sep 17 00:00:00 2001 From: Ayush Patel Date: Mon, 21 Sep 2026 15:07:18 +0000 Subject: [PATCH 1/3] feat: allow omitting reasoning effort so the model default applies --- src/benchmarks/benchmark-config.test.ts | 5 +++-- src/benchmarks/benchmark-config.ts | 2 +- src/benchmarks/search/core/request.test.ts | 4 ++++ src/benchmarks/search/core/request.ts | 7 +++++-- src/benchmarks/search/core/solver.ts | 2 +- src/cli/index.test.ts | 22 ++++++++++++++++++++++ src/cli/index.ts | 20 ++++++++++++++------ src/harness/model.ts | 2 +- src/providers/responses-model.test.ts | 1 + src/providers/responses-model.ts | 5 ++++- 10 files changed, 56 insertions(+), 14 deletions(-) diff --git a/src/benchmarks/benchmark-config.test.ts b/src/benchmarks/benchmark-config.test.ts index 4cba901a..5aa74440 100644 --- a/src/benchmarks/benchmark-config.test.ts +++ b/src/benchmarks/benchmark-config.test.ts @@ -38,13 +38,14 @@ describe("benchmark config", () => { assertLeft(result); }); - it("requires reasoningEffort in model benchmark configs", () => { + it("accepts model benchmark configs without reasoningEffort", () => { const result = parseSchema(BenchmarkRunConfigSchema, { benchmarkId: "gpqa_diamond", model: "openai/gpt-5", }); - assertLeft(result); + assertRight(result); + expect(result.right).not.toHaveProperty("reasoningEffort"); }); it("parses injected benchmark configs with opaque options", () => { diff --git a/src/benchmarks/benchmark-config.ts b/src/benchmarks/benchmark-config.ts index 24f309ea..668676f3 100644 --- a/src/benchmarks/benchmark-config.ts +++ b/src/benchmarks/benchmark-config.ts @@ -44,7 +44,7 @@ export type GeminiMediaResolution = ValueOf; export const InferenceOverrideSchema = z.object({ temperature: z.number().optional(), maxTokens: z.number().optional(), - reasoningEffort: z.enum(REASONING_EFFORTS), + reasoningEffort: z.enum(REASONING_EFFORTS).optional(), costTier: z.enum(COST_TIERS).optional(), timeoutMs: z.number().optional(), sort: z.nativeEnum(ProviderSort).optional(), diff --git a/src/benchmarks/search/core/request.test.ts b/src/benchmarks/search/core/request.test.ts index da276ff8..60b349b5 100644 --- a/src/benchmarks/search/core/request.test.ts +++ b/src/benchmarks/search/core/request.test.ts @@ -196,6 +196,10 @@ describe("buildSearchRequestBody", () => { expect(body.temperature).toBe(0); expect(body.reasoning).toEqual({ effort: "high" }); }); + it("omits reasoning when no effort is requested", () => { + const body = buildSearchRequestBody({ ...BASE, lane: lane({}) }); + expect(body).not.toHaveProperty("reasoning"); + }); it("threads the candidate models list and omits it by default", () => { const routed = buildSearchRequestBody({ ...BASE, diff --git a/src/benchmarks/search/core/request.ts b/src/benchmarks/search/core/request.ts index c2cf6588..aec4730b 100644 --- a/src/benchmarks/search/core/request.ts +++ b/src/benchmarks/search/core/request.ts @@ -28,7 +28,7 @@ export interface SearchRequestOptions { readonly lane: SearchLaneConfig; readonly maxOutputTokens?: number; readonly temperature?: number; - readonly reasoningEffort: ReasoningEffort; + readonly reasoningEffort?: ReasoningEffort; readonly sort?: ProviderSort; readonly providerOrder?: readonly string[]; readonly providerOnly?: readonly string[]; @@ -127,7 +127,10 @@ export function buildSearchRequestBody( models: opts.models === undefined ? undefined : [...opts.models], maxOutputTokens: opts.maxOutputTokens, temperature: opts.temperature, - reasoning: { effort: opts.reasoningEffort }, + reasoning: + opts.reasoningEffort !== undefined + ? { effort: opts.reasoningEffort } + : undefined, provider: opts.sort !== undefined || opts.providerOrder !== undefined || diff --git a/src/benchmarks/search/core/solver.ts b/src/benchmarks/search/core/solver.ts index f9aae09a..2adfbd28 100644 --- a/src/benchmarks/search/core/solver.ts +++ b/src/benchmarks/search/core/solver.ts @@ -50,7 +50,7 @@ export interface SearchSolverOptions { readonly timeoutMs?: number; readonly maxOutputTokens?: number; readonly temperature?: number; - readonly reasoningEffort: ReasoningEffort; + readonly reasoningEffort?: ReasoningEffort; readonly endpointId?: string; readonly sort?: ProviderSort; readonly providerOrder?: readonly string[]; diff --git a/src/cli/index.test.ts b/src/cli/index.test.ts index 326bff7b..5d20a922 100644 --- a/src/cli/index.test.ts +++ b/src/cli/index.test.ts @@ -12,6 +12,28 @@ describe("bench-harness CLI", () => { ); }); + it("leaves reasoning effort unset for --reasoning-effort default", () => { + const args = parseArgs([ + "--benchmark", + "gpqa_diamond", + "--model", + "openrouter/jev", + "--reasoning-effort", + "default", + ]); + expect(args.reasoningEffort).toBeUndefined(); + const config = buildBenchmarkConfig({ + benchmarkId: "gpqa_diamond", + model: "openrouter/jev", + panelConfig: undefined, + artifactDir: undefined, + endpointId: undefined, + imageDetail: undefined, + reasoningEffort: args.reasoningEffort, + }); + expect(config).not.toHaveProperty("reasoningEffort"); + }); + it("rejects an invalid --reasoning-effort value", () => { expect(() => parseArgs(["--reasoning-effort", "invalid"])).toThrow( "--reasoning-effort must be one of" diff --git a/src/cli/index.ts b/src/cli/index.ts index 4c9aea8d..f04cec47 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -46,7 +46,7 @@ interface CliArgs { readonly resumeId?: string; readonly imageDetail?: ImageDetail; readonly costTier?: CostTier; - readonly reasoningEffort: ReasoningEffort; + readonly reasoningEffort: ReasoningEffort | undefined; } export function parseArgs(argv: readonly string[]): CliArgs { @@ -190,7 +190,7 @@ function main(): Promise { reasoningEffort: args.reasoningEffort, }); process.stderr.write( - `Running ${args.benchmark}${args.model !== undefined ? ` on ${args.model}` : ""}${args.solverConfig !== undefined ? ` (solver-config=${args.solverConfig})` : ""}${artifactDir !== undefined ? ` (artifact-dir=${artifactDir})` : ""} (epochs=${epochs}, concurrency=${args.concurrency}, reasoning-effort=${args.reasoningEffort}${range !== undefined ? `, range=${range.start ?? 0}..${range.end ?? "end"}` : ""}, session=${sessionId})...\n` + `Running ${args.benchmark}${args.model !== undefined ? ` on ${args.model}` : ""}${args.solverConfig !== undefined ? ` (solver-config=${args.solverConfig})` : ""}${artifactDir !== undefined ? ` (artifact-dir=${artifactDir})` : ""} (epochs=${epochs}, concurrency=${args.concurrency}, reasoning-effort=${args.reasoningEffort ?? MODEL_DEFAULT_REASONING_EFFORT}${range !== undefined ? `, range=${range.start ?? 0}..${range.end ?? "end"}` : ""}, session=${sessionId})...\n` ); const total = yield* promise(() => resolveTotalEvaluations(args.benchmark, range, epochs) @@ -304,13 +304,20 @@ function validateCostTier(raw: string | undefined): CostTier | undefined { return raw; } -function validateReasoningEffort(raw: string | undefined): ReasoningEffort { +const MODEL_DEFAULT_REASONING_EFFORT = "default"; + +function validateReasoningEffort( + raw: string | undefined +): ReasoningEffort | undefined { if (raw === undefined) { return DEFAULT_REASONING_EFFORT; } + if (raw === MODEL_DEFAULT_REASONING_EFFORT) { + return undefined; + } if (!isMember(raw, REASONING_EFFORTS)) { throw new Error( - `--reasoning-effort must be one of: ${REASONING_EFFORTS.join(", ")} (got "${raw}")` + `--reasoning-effort must be one of: ${[...REASONING_EFFORTS, MODEL_DEFAULT_REASONING_EFFORT].join(", ")} (got "${raw}")` ); } return raw; @@ -330,7 +337,7 @@ function buildSchemaValidatedConfig(opts: { panelConfig: unknown; imageDetail?: ImageDetail; costTier?: CostTier; - reasoningEffort: ReasoningEffort; + reasoningEffort: ReasoningEffort | undefined; }): BenchmarkRunConfig { const { benchmarkId, @@ -372,6 +379,7 @@ function buildSchemaValidatedConfig(opts: { : undefined; if ( optionsSchema !== undefined && + reasoningEffort !== undefined && Object.hasOwn(optionsSchema.shape, "agentReasoningEffort") && !( typeof panelConfig === "object" && @@ -410,7 +418,7 @@ export function buildBenchmarkConfig(opts: { endpointId: string | undefined; imageDetail: ImageDetail | undefined; costTier?: CostTier; - reasoningEffort: ReasoningEffort; + reasoningEffort: ReasoningEffort | undefined; }): BenchmarkRunConfig { const { benchmarkId, diff --git a/src/harness/model.ts b/src/harness/model.ts index 0e9d4b5f..0f85c282 100644 --- a/src/harness/model.ts +++ b/src/harness/model.ts @@ -20,7 +20,7 @@ export interface GenerateConfig { readonly maxTokens?: number; readonly endpointId?: string; readonly tools?: readonly ToolDefinition[]; - readonly reasoningEffort: ReasoningEffort; + readonly reasoningEffort?: ReasoningEffort; readonly costTier?: CostTier; readonly timeoutMs?: number; readonly sort?: ProviderSort; diff --git a/src/providers/responses-model.test.ts b/src/providers/responses-model.test.ts index e3dd04e2..ec840278 100644 --- a/src/providers/responses-model.test.ts +++ b/src/providers/responses-model.test.ts @@ -313,6 +313,7 @@ describe("responses-model", () => { }, ]); expect(sentOptions?.extraBody).toBeUndefined(); + expect(sentBody).not.toHaveProperty("reasoning"); expect(exit.value.outputItems).toEqual([ { type: "function_call", diff --git a/src/providers/responses-model.ts b/src/providers/responses-model.ts index b558d63d..81293e59 100644 --- a/src/providers/responses-model.ts +++ b/src/providers/responses-model.ts @@ -185,8 +185,11 @@ export function generate( ? [...genConfig.tools] : undefined, }), - reasoning: { effort: genConfig.reasoningEffort }, ...definedValues({ + reasoning: + genConfig.reasoningEffort !== undefined + ? { effort: genConfig.reasoningEffort } + : undefined, provider: sendProvider ? providerPreferences : undefined, }), ...definedValues({ From c3bf9ce2132fdc116fc16c0de9fc9bf9ff8f539d Mon Sep 17 00:00:00 2001 From: Ayush Patel Date: Mon, 21 Sep 2026 15:12:41 +0000 Subject: [PATCH 2/3] feat: default to adaptive reasoning effort for openrouter/jev --- src/cli/index.test.ts | 14 ++++++++++++++ src/cli/index.ts | 12 ++++++++---- src/harness/constants.ts | 12 ++++++++++++ 3 files changed, 34 insertions(+), 4 deletions(-) diff --git a/src/cli/index.test.ts b/src/cli/index.test.ts index 5d20a922..5ccfea6b 100644 --- a/src/cli/index.test.ts +++ b/src/cli/index.test.ts @@ -12,6 +12,20 @@ describe("bench-harness CLI", () => { ); }); + it("leaves reasoning effort unset by default for adaptive-effort routers", () => { + expect( + parseArgs(["--model", "openrouter/jev"]).reasoningEffort + ).toBeUndefined(); + expect( + parseArgs(["--model", "openrouter/jev:nitro"]).reasoningEffort + ).toBeUndefined(); + expect(parseArgs(["--model", "openai/gpt-5"]).reasoningEffort).toBe("high"); + expect( + parseArgs(["--model", "openrouter/jev", "--reasoning-effort", "high"]) + .reasoningEffort + ).toBe("high"); + }); + it("leaves reasoning effort unset for --reasoning-effort default", () => { const args = parseArgs([ "--benchmark", diff --git a/src/cli/index.ts b/src/cli/index.ts index f04cec47..e891e339 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -19,7 +19,7 @@ import { benchmarkIds, getBenchmark } from "../benchmarks/registry"; import type { CostTier, ReasoningEffort } from "../harness/constants"; import { COST_TIERS, - DEFAULT_REASONING_EFFORT, + defaultReasoningEffortFor, ImageDetail, IMAGE_DETAIL_VALUES, REASONING_EFFORTS, @@ -72,7 +72,10 @@ export function parseArgs(argv: readonly string[]): CliArgs { resumeId: get("--resume-id"), imageDetail: validateImageDetail(get("--image-detail")), costTier: validateCostTier(get("--cost-tier")), - reasoningEffort: validateReasoningEffort(get("--reasoning-effort")), + reasoningEffort: validateReasoningEffort( + get("--reasoning-effort"), + get("--model") + ), }; } @@ -307,10 +310,11 @@ function validateCostTier(raw: string | undefined): CostTier | undefined { const MODEL_DEFAULT_REASONING_EFFORT = "default"; function validateReasoningEffort( - raw: string | undefined + raw: string | undefined, + model: string | undefined ): ReasoningEffort | undefined { if (raw === undefined) { - return DEFAULT_REASONING_EFFORT; + return defaultReasoningEffortFor(model); } if (raw === MODEL_DEFAULT_REASONING_EFFORT) { return undefined; diff --git a/src/harness/constants.ts b/src/harness/constants.ts index f6bf2ec8..fe8357e7 100644 --- a/src/harness/constants.ts +++ b/src/harness/constants.ts @@ -13,6 +13,18 @@ export type ReasoningEffort = ValueOf; export const DEFAULT_REASONING_EFFORT: ReasoningEffort = "high"; +export const ADAPTIVE_REASONING_EFFORT_MODELS = ["openrouter/jev"] as const; + +export function defaultReasoningEffortFor( + model: string | undefined +): ReasoningEffort | undefined { + const baseModel = model?.split(":")[0]; + return baseModel !== undefined && + ADAPTIVE_REASONING_EFFORT_MODELS.some((slug) => slug === baseModel) + ? undefined + : DEFAULT_REASONING_EFFORT; +} + export const COST_TIERS = ["low", "medium", "high", "xhigh", "max"] as const; export type CostTier = ValueOf; From b837ddd1eb73aea87a8303a9d4c789e1b71fe41c Mon Sep 17 00:00:00 2001 From: Ayush Patel Date: Mon, 21 Sep 2026 15:36:57 +0000 Subject: [PATCH 3/3] feat: add explicit auto reasoning effort for openrouter/jev reasoningEffort stays required in every benchmark config. The new auto value is the default for openrouter/jev (and its variants) and maps to no reasoning object on the wire, so the router's adaptive effort selection applies. Concrete efforts are still pinned and serialized. auto is rejected by the CLI for models without adaptive effort and is never forwarded as an ori agent reasoning effort. --- src/benchmarks/agent-cli/schema.ts | 8 +++ src/benchmarks/benchmark-config.test.ts | 14 +++- src/benchmarks/benchmark-config.ts | 2 +- src/benchmarks/deep-swe/solver.ts | 26 ++++--- src/benchmarks/search/core/request.test.ts | 10 ++- src/benchmarks/search/core/request.ts | 8 +-- src/benchmarks/search/core/solver.ts | 2 +- src/benchmarks/swe-atlas/solver.ts | 24 +++++-- src/cli/index.test.ts | 84 +++++++++++++--------- src/cli/index.ts | 30 ++++---- src/harness/constants.test.ts | 18 +++++ src/harness/constants.ts | 30 ++++++-- src/harness/model.ts | 2 +- src/judge/judge.ts | 3 +- src/providers/responses-model.test.ts | 21 +++++- src/providers/responses-model.ts | 6 +- 16 files changed, 203 insertions(+), 85 deletions(-) create mode 100644 src/harness/constants.test.ts diff --git a/src/benchmarks/agent-cli/schema.ts b/src/benchmarks/agent-cli/schema.ts index 57fc926c..bdd69e91 100644 --- a/src/benchmarks/agent-cli/schema.ts +++ b/src/benchmarks/agent-cli/schema.ts @@ -1,3 +1,5 @@ +import type { ReasoningEffort } from "../../harness/constants"; +import { ADAPTIVE_REASONING_EFFORT } from "../../harness/constants"; import type { ValueOf } from "../../internal/guards"; export const ORI_AGENTS = ["pi", "claude", "prime-agent", "omp"] as const; @@ -22,6 +24,12 @@ export const ORI_REASONING_EFFORTS = [ export type OriReasoningEffort = ValueOf; +export function toOriReasoningEffort( + effort: ReasoningEffort +): OriReasoningEffort | undefined { + return effort === ADAPTIVE_REASONING_EFFORT ? undefined : effort; +} + export const AGENT_PACKAGE_PATTERN = /^[A-Za-z0-9@/:._^=+-]+$/; export function isValidAgentPackage(value: string): boolean { diff --git a/src/benchmarks/benchmark-config.test.ts b/src/benchmarks/benchmark-config.test.ts index 5aa74440..4405e1f3 100644 --- a/src/benchmarks/benchmark-config.test.ts +++ b/src/benchmarks/benchmark-config.test.ts @@ -38,14 +38,24 @@ describe("benchmark config", () => { assertLeft(result); }); - it("accepts model benchmark configs without reasoningEffort", () => { + it("requires reasoningEffort in model benchmark configs", () => { const result = parseSchema(BenchmarkRunConfigSchema, { benchmarkId: "gpqa_diamond", model: "openai/gpt-5", }); + assertLeft(result); + }); + + it("accepts the explicit auto reasoning effort", () => { + const result = parseSchema(BenchmarkRunConfigSchema, { + benchmarkId: "gpqa_diamond", + model: "openrouter/jev", + reasoningEffort: "auto", + }); + assertRight(result); - expect(result.right).not.toHaveProperty("reasoningEffort"); + expect(result.right.reasoningEffort).toBe("auto"); }); it("parses injected benchmark configs with opaque options", () => { diff --git a/src/benchmarks/benchmark-config.ts b/src/benchmarks/benchmark-config.ts index 668676f3..24f309ea 100644 --- a/src/benchmarks/benchmark-config.ts +++ b/src/benchmarks/benchmark-config.ts @@ -44,7 +44,7 @@ export type GeminiMediaResolution = ValueOf; export const InferenceOverrideSchema = z.object({ temperature: z.number().optional(), maxTokens: z.number().optional(), - reasoningEffort: z.enum(REASONING_EFFORTS).optional(), + reasoningEffort: z.enum(REASONING_EFFORTS), costTier: z.enum(COST_TIERS).optional(), timeoutMs: z.number().optional(), sort: z.nativeEnum(ProviderSort).optional(), diff --git a/src/benchmarks/deep-swe/solver.ts b/src/benchmarks/deep-swe/solver.ts index fa8bca00..469f5abf 100644 --- a/src/benchmarks/deep-swe/solver.ts +++ b/src/benchmarks/deep-swe/solver.ts @@ -39,7 +39,7 @@ import { runAgentCli, } from "../agent-cli/runner"; import type { HarborAgent } from "../agent-cli/schema"; -import { isOriAgent } from "../agent-cli/schema"; +import { isOriAgent, toOriReasoningEffort } from "../agent-cli/schema"; import type { InferenceOverride } from "../benchmark-config"; import type { AgentLoopInput } from "../harbor/agent-loop"; import { AGENT_ENV, probeSystemInfo, runAgentLoop } from "../harbor/agent-loop"; @@ -104,15 +104,25 @@ export function makeDeepSweSolver( const task = loadTask(meta.taskId, tasksRoot); const agent = opts.agent ?? "mini_swe"; const cliHarness = isOriAgent(agent) ? getOriHarness(agent) : undefined; - const baseCliOpts: AgentCliOpts = + const agentReasoningEffort = toOriReasoningEffort( + opts.inference.reasoningEffort + ); + const baseCliOpts: AgentCliOpts | undefined = opts.agentCli ?? - definedValues({ - model: opts.model, - apiKey: opts.apiKey, - endpointId: opts.endpointId, - sessionId: opts.sessionId, - agentReasoningEffort: opts.inference.reasoningEffort, + (agentReasoningEffort === undefined + ? undefined + : definedValues({ + model: opts.model, + apiKey: opts.apiKey, + endpointId: opts.endpointId, + sessionId: opts.sessionId, + agentReasoningEffort, + })); + if (baseCliOpts === undefined) { + return yield* new SolverError({ + message: `deep-swe agent CLI requires a pinned reasoning effort (got "${opts.inference.reasoningEffort}")`, }); + } const cliOpts: AgentCliOpts = { ...baseCliOpts, appendSystemPrompt: diff --git a/src/benchmarks/search/core/request.test.ts b/src/benchmarks/search/core/request.test.ts index 60b349b5..30aed2fb 100644 --- a/src/benchmarks/search/core/request.test.ts +++ b/src/benchmarks/search/core/request.test.ts @@ -196,9 +196,13 @@ describe("buildSearchRequestBody", () => { expect(body.temperature).toBe(0); expect(body.reasoning).toEqual({ effort: "high" }); }); - it("omits reasoning when no effort is requested", () => { - const body = buildSearchRequestBody({ ...BASE, lane: lane({}) }); - expect(body).not.toHaveProperty("reasoning"); + it("omits reasoning on the wire when effort is auto", () => { + const body = buildSearchRequestBody({ + ...BASE, + lane: lane({}), + reasoningEffort: "auto", + }); + expect(Object.hasOwn(body, "reasoning")).toBe(false); }); it("threads the candidate models list and omits it by default", () => { const routed = buildSearchRequestBody({ diff --git a/src/benchmarks/search/core/request.ts b/src/benchmarks/search/core/request.ts index aec4730b..8362daa3 100644 --- a/src/benchmarks/search/core/request.ts +++ b/src/benchmarks/search/core/request.ts @@ -9,6 +9,7 @@ import type { } from "@openrouter/sdk/models"; import type { CostTier, ReasoningEffort } from "../../../harness/constants"; +import { reasoningRequestFor } from "../../../harness/constants"; import type { ProviderSort } from "../../../internal/enums"; import { definedValues } from "../../../internal/guards"; import { buildAutoRouterPlugin } from "../../../providers/auto-router-plugin"; @@ -28,7 +29,7 @@ export interface SearchRequestOptions { readonly lane: SearchLaneConfig; readonly maxOutputTokens?: number; readonly temperature?: number; - readonly reasoningEffort?: ReasoningEffort; + readonly reasoningEffort: ReasoningEffort; readonly sort?: ProviderSort; readonly providerOrder?: readonly string[]; readonly providerOnly?: readonly string[]; @@ -127,10 +128,7 @@ export function buildSearchRequestBody( models: opts.models === undefined ? undefined : [...opts.models], maxOutputTokens: opts.maxOutputTokens, temperature: opts.temperature, - reasoning: - opts.reasoningEffort !== undefined - ? { effort: opts.reasoningEffort } - : undefined, + reasoning: reasoningRequestFor(opts.reasoningEffort), provider: opts.sort !== undefined || opts.providerOrder !== undefined || diff --git a/src/benchmarks/search/core/solver.ts b/src/benchmarks/search/core/solver.ts index 2adfbd28..f9aae09a 100644 --- a/src/benchmarks/search/core/solver.ts +++ b/src/benchmarks/search/core/solver.ts @@ -50,7 +50,7 @@ export interface SearchSolverOptions { readonly timeoutMs?: number; readonly maxOutputTokens?: number; readonly temperature?: number; - readonly reasoningEffort?: ReasoningEffort; + readonly reasoningEffort: ReasoningEffort; readonly endpointId?: string; readonly sort?: ProviderSort; readonly providerOrder?: readonly string[]; diff --git a/src/benchmarks/swe-atlas/solver.ts b/src/benchmarks/swe-atlas/solver.ts index a65f2e5a..38d1ceea 100644 --- a/src/benchmarks/swe-atlas/solver.ts +++ b/src/benchmarks/swe-atlas/solver.ts @@ -23,7 +23,7 @@ import { runAgentCli, } from "../agent-cli/runner"; import type { HarborAgent } from "../agent-cli/schema"; -import { isOriAgent } from "../agent-cli/schema"; +import { isOriAgent, toOriReasoningEffort } from "../agent-cli/schema"; import type { InferenceOverride } from "../benchmark-config"; import { AGENT_ENV, probeSystemInfo, runAgentLoop } from "../harbor/agent-loop"; import { @@ -105,14 +105,24 @@ export function makeSweAtlasSolver( const task = loadTask(meta.taskId, meta.track, tasksRoot); const agent = opts.agent ?? "mini_swe"; const cliHarness = isOriAgent(agent) ? getOriHarness(agent) : undefined; - const baseCliOpts: AgentCliOpts = + const agentReasoningEffort = toOriReasoningEffort( + opts.inference.reasoningEffort + ); + const baseCliOpts: AgentCliOpts | undefined = opts.agentCli ?? - definedValues({ - model: opts.model, - apiKey: opts.apiKey, - endpointId: opts.endpointId, - agentReasoningEffort: opts.inference.reasoningEffort, + (agentReasoningEffort === undefined + ? undefined + : definedValues({ + model: opts.model, + apiKey: opts.apiKey, + endpointId: opts.endpointId, + agentReasoningEffort, + })); + if (baseCliOpts === undefined) { + return yield* new SolverError({ + message: `swe-atlas agent CLI requires a pinned reasoning effort (got "${opts.inference.reasoningEffort}")`, }); + } const cliOpts: AgentCliOpts = { ...baseCliOpts, appendSystemPrompt: joinAgentPrompts( diff --git a/src/cli/index.test.ts b/src/cli/index.test.ts index 5ccfea6b..11086512 100644 --- a/src/cli/index.test.ts +++ b/src/cli/index.test.ts @@ -4,48 +4,39 @@ import { buildBenchmarkConfig, parseArgs } from "."; describe("bench-harness CLI", () => { it("defaults --reasoning-effort to high", () => { expect(parseArgs([]).reasoningEffort).toBe("high"); + expect(parseArgs(["--model", "openai/gpt-5"]).reasoningEffort).toBe("high"); }); - it("accepts an explicit --reasoning-effort", () => { - expect(parseArgs(["--reasoning-effort", "low"]).reasoningEffort).toBe( - "low" + it("defaults --reasoning-effort to auto for openrouter/jev", () => { + expect(parseArgs(["--model", "openrouter/jev"]).reasoningEffort).toBe( + "auto" + ); + expect(parseArgs(["--model", "openrouter/jev:nitro"]).reasoningEffort).toBe( + "auto" ); - }); - - it("leaves reasoning effort unset by default for adaptive-effort routers", () => { - expect( - parseArgs(["--model", "openrouter/jev"]).reasoningEffort - ).toBeUndefined(); - expect( - parseArgs(["--model", "openrouter/jev:nitro"]).reasoningEffort - ).toBeUndefined(); - expect(parseArgs(["--model", "openai/gpt-5"]).reasoningEffort).toBe("high"); expect( parseArgs(["--model", "openrouter/jev", "--reasoning-effort", "high"]) .reasoningEffort ).toBe("high"); + expect( + parseArgs(["--model", "openrouter/jev", "--reasoning-effort", "auto"]) + .reasoningEffort + ).toBe("auto"); }); - it("leaves reasoning effort unset for --reasoning-effort default", () => { - const args = parseArgs([ - "--benchmark", - "gpqa_diamond", - "--model", - "openrouter/jev", - "--reasoning-effort", - "default", - ]); - expect(args.reasoningEffort).toBeUndefined(); - const config = buildBenchmarkConfig({ - benchmarkId: "gpqa_diamond", - model: "openrouter/jev", - panelConfig: undefined, - artifactDir: undefined, - endpointId: undefined, - imageDetail: undefined, - reasoningEffort: args.reasoningEffort, - }); - expect(config).not.toHaveProperty("reasoningEffort"); + it("rejects --reasoning-effort auto for models without adaptive effort", () => { + expect(() => + parseArgs(["--model", "openai/gpt-5", "--reasoning-effort", "auto"]) + ).toThrow("--reasoning-effort auto is only supported for"); + expect(() => parseArgs(["--reasoning-effort", "auto"])).toThrow( + "--reasoning-effort auto is only supported for" + ); + }); + + it("accepts an explicit --reasoning-effort", () => { + expect(parseArgs(["--reasoning-effort", "low"]).reasoningEffort).toBe( + "low" + ); }); it("rejects an invalid --reasoning-effort value", () => { @@ -177,6 +168,33 @@ describe("bench-harness CLI", () => { }); }); + it("does not derive an ori agent reasoning effort from auto", () => { + expect(() => + buildBenchmarkConfig({ + benchmarkId: "terminal_bench", + model: "openrouter/jev", + panelConfig: undefined, + artifactDir: undefined, + endpointId: undefined, + imageDetail: undefined, + reasoningEffort: "auto", + }) + ).toThrow(); + const config = buildBenchmarkConfig({ + benchmarkId: "terminal_bench", + model: "openrouter/jev", + panelConfig: { agentReasoningEffort: "high" }, + artifactDir: undefined, + endpointId: undefined, + imageDetail: undefined, + reasoningEffort: "auto", + }); + expect(config).toMatchObject({ + reasoningEffort: "auto", + agentReasoningEffort: "high", + }); + }); + it("preserves explicit ori agent reasoning effort", () => { const config = buildBenchmarkConfig({ benchmarkId: "terminal_bench", diff --git a/src/cli/index.ts b/src/cli/index.ts index e891e339..cf8487cf 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -18,11 +18,14 @@ import { DracoPanelConfigSchema } from "../benchmarks/draco/schemas"; import { benchmarkIds, getBenchmark } from "../benchmarks/registry"; import type { CostTier, ReasoningEffort } from "../harness/constants"; import { + ADAPTIVE_REASONING_EFFORT, + ADAPTIVE_REASONING_EFFORT_MODELS, COST_TIERS, defaultReasoningEffortFor, ImageDetail, IMAGE_DETAIL_VALUES, REASONING_EFFORTS, + supportsAdaptiveReasoningEffort, } from "../harness/constants"; import { makeProgressReporter } from "../harness/progress"; import { runHarnessPromise } from "../internal/effect-logger"; @@ -46,7 +49,7 @@ interface CliArgs { readonly resumeId?: string; readonly imageDetail?: ImageDetail; readonly costTier?: CostTier; - readonly reasoningEffort: ReasoningEffort | undefined; + readonly reasoningEffort: ReasoningEffort; } export function parseArgs(argv: readonly string[]): CliArgs { @@ -193,7 +196,7 @@ function main(): Promise { reasoningEffort: args.reasoningEffort, }); process.stderr.write( - `Running ${args.benchmark}${args.model !== undefined ? ` on ${args.model}` : ""}${args.solverConfig !== undefined ? ` (solver-config=${args.solverConfig})` : ""}${artifactDir !== undefined ? ` (artifact-dir=${artifactDir})` : ""} (epochs=${epochs}, concurrency=${args.concurrency}, reasoning-effort=${args.reasoningEffort ?? MODEL_DEFAULT_REASONING_EFFORT}${range !== undefined ? `, range=${range.start ?? 0}..${range.end ?? "end"}` : ""}, session=${sessionId})...\n` + `Running ${args.benchmark}${args.model !== undefined ? ` on ${args.model}` : ""}${args.solverConfig !== undefined ? ` (solver-config=${args.solverConfig})` : ""}${artifactDir !== undefined ? ` (artifact-dir=${artifactDir})` : ""} (epochs=${epochs}, concurrency=${args.concurrency}, reasoning-effort=${args.reasoningEffort}${range !== undefined ? `, range=${range.start ?? 0}..${range.end ?? "end"}` : ""}, session=${sessionId})...\n` ); const total = yield* promise(() => resolveTotalEvaluations(args.benchmark, range, epochs) @@ -307,21 +310,24 @@ function validateCostTier(raw: string | undefined): CostTier | undefined { return raw; } -const MODEL_DEFAULT_REASONING_EFFORT = "default"; - function validateReasoningEffort( raw: string | undefined, model: string | undefined -): ReasoningEffort | undefined { +): ReasoningEffort { if (raw === undefined) { return defaultReasoningEffortFor(model); } - if (raw === MODEL_DEFAULT_REASONING_EFFORT) { - return undefined; - } if (!isMember(raw, REASONING_EFFORTS)) { throw new Error( - `--reasoning-effort must be one of: ${[...REASONING_EFFORTS, MODEL_DEFAULT_REASONING_EFFORT].join(", ")} (got "${raw}")` + `--reasoning-effort must be one of: ${REASONING_EFFORTS.join(", ")} (got "${raw}")` + ); + } + if ( + raw === ADAPTIVE_REASONING_EFFORT && + (model === undefined || !supportsAdaptiveReasoningEffort(model)) + ) { + throw new Error( + `--reasoning-effort ${ADAPTIVE_REASONING_EFFORT} is only supported for: ${ADAPTIVE_REASONING_EFFORT_MODELS.join(", ")} (got model "${model ?? ""}")` ); } return raw; @@ -341,7 +347,7 @@ function buildSchemaValidatedConfig(opts: { panelConfig: unknown; imageDetail?: ImageDetail; costTier?: CostTier; - reasoningEffort: ReasoningEffort | undefined; + reasoningEffort: ReasoningEffort; }): BenchmarkRunConfig { const { benchmarkId, @@ -383,7 +389,7 @@ function buildSchemaValidatedConfig(opts: { : undefined; if ( optionsSchema !== undefined && - reasoningEffort !== undefined && + reasoningEffort !== ADAPTIVE_REASONING_EFFORT && Object.hasOwn(optionsSchema.shape, "agentReasoningEffort") && !( typeof panelConfig === "object" && @@ -422,7 +428,7 @@ export function buildBenchmarkConfig(opts: { endpointId: string | undefined; imageDetail: ImageDetail | undefined; costTier?: CostTier; - reasoningEffort: ReasoningEffort | undefined; + reasoningEffort: ReasoningEffort; }): BenchmarkRunConfig { const { benchmarkId, diff --git a/src/harness/constants.test.ts b/src/harness/constants.test.ts new file mode 100644 index 00000000..ca6a2e0a --- /dev/null +++ b/src/harness/constants.test.ts @@ -0,0 +1,18 @@ +import { describe, expect, it } from "bun:test"; + +import { defaultReasoningEffortFor, reasoningRequestFor } from "./constants"; + +describe("reasoning effort constants", () => { + it("defaults openrouter/jev to auto and everything else to high", () => { + expect(defaultReasoningEffortFor("openrouter/jev")).toBe("auto"); + expect(defaultReasoningEffortFor("openrouter/jev:nitro")).toBe("auto"); + expect(defaultReasoningEffortFor("openai/gpt-5")).toBe("high"); + expect(defaultReasoningEffortFor(undefined)).toBe("high"); + }); + + it("maps auto to no reasoning object and pins every other effort", () => { + expect(reasoningRequestFor("auto")).toBeUndefined(); + expect(reasoningRequestFor("high")).toEqual({ effort: "high" }); + expect(reasoningRequestFor("none")).toEqual({ effort: "none" }); + }); +}); diff --git a/src/harness/constants.ts b/src/harness/constants.ts index fe8357e7..614d545f 100644 --- a/src/harness/constants.ts +++ b/src/harness/constants.ts @@ -1,6 +1,8 @@ import type { ValueOf } from "../internal/guards"; -export const REASONING_EFFORTS = [ +export const ADAPTIVE_REASONING_EFFORT = "auto"; + +export const PINNED_REASONING_EFFORTS = [ "xhigh", "high", "medium", @@ -9,22 +11,38 @@ export const REASONING_EFFORTS = [ "none", ] as const; +export type PinnedReasoningEffort = ValueOf; + +export const REASONING_EFFORTS = [ + ...PINNED_REASONING_EFFORTS, + ADAPTIVE_REASONING_EFFORT, +] as const; + export type ReasoningEffort = ValueOf; export const DEFAULT_REASONING_EFFORT: ReasoningEffort = "high"; export const ADAPTIVE_REASONING_EFFORT_MODELS = ["openrouter/jev"] as const; +export function supportsAdaptiveReasoningEffort(model: string): boolean { + const baseModel = model.split(":")[0]; + return ADAPTIVE_REASONING_EFFORT_MODELS.some((slug) => slug === baseModel); +} + export function defaultReasoningEffortFor( model: string | undefined -): ReasoningEffort | undefined { - const baseModel = model?.split(":")[0]; - return baseModel !== undefined && - ADAPTIVE_REASONING_EFFORT_MODELS.some((slug) => slug === baseModel) - ? undefined +): ReasoningEffort { + return model !== undefined && supportsAdaptiveReasoningEffort(model) + ? ADAPTIVE_REASONING_EFFORT : DEFAULT_REASONING_EFFORT; } +export function reasoningRequestFor( + effort: ReasoningEffort +): { readonly effort: PinnedReasoningEffort } | undefined { + return effort === ADAPTIVE_REASONING_EFFORT ? undefined : { effort }; +} + export const COST_TIERS = ["low", "medium", "high", "xhigh", "max"] as const; export type CostTier = ValueOf; diff --git a/src/harness/model.ts b/src/harness/model.ts index 0f85c282..0e9d4b5f 100644 --- a/src/harness/model.ts +++ b/src/harness/model.ts @@ -20,7 +20,7 @@ export interface GenerateConfig { readonly maxTokens?: number; readonly endpointId?: string; readonly tools?: readonly ToolDefinition[]; - readonly reasoningEffort?: ReasoningEffort; + readonly reasoningEffort: ReasoningEffort; readonly costTier?: CostTier; readonly timeoutMs?: number; readonly sort?: ProviderSort; diff --git a/src/judge/judge.ts b/src/judge/judge.ts index 89499dd6..f134ffa8 100644 --- a/src/judge/judge.ts +++ b/src/judge/judge.ts @@ -6,6 +6,7 @@ import type { Effect } from "effect/Effect"; import { fail, flatMap, mapError, succeed } from "effect/Effect"; import type { ReasoningEffort } from "../harness/constants"; +import { reasoningRequestFor } from "../harness/constants"; import type { ModelUsage } from "../harness/core"; import { ModelError } from "../harness/core"; import { Either } from "../internal/either"; @@ -67,7 +68,7 @@ export function judgeCall( text, instructions: spec.instructions, temperature: config.temperature, - reasoning: { effort: config.reasoningEffort }, + reasoning: reasoningRequestFor(config.reasoningEffort), }); const sendOptions = definedValues({ timeoutMs: config.timeoutMs ?? DEFAULT_JUDGE_TIMEOUT_MS, diff --git a/src/providers/responses-model.test.ts b/src/providers/responses-model.test.ts index ec840278..117ef46a 100644 --- a/src/providers/responses-model.test.ts +++ b/src/providers/responses-model.test.ts @@ -313,7 +313,6 @@ describe("responses-model", () => { }, ]); expect(sentOptions?.extraBody).toBeUndefined(); - expect(sentBody).not.toHaveProperty("reasoning"); expect(exit.value.outputItems).toEqual([ { type: "function_call", @@ -361,6 +360,26 @@ describe("responses-model", () => { expect(captured.value?.body["provider"]).toBeUndefined(); expect(captured.value?.headers["x-or-endpoint-id"]).toBe("endpoint-1"); }); + it("omits reasoning on the wire when effort is auto", async () => { + const captured: { + value: CapturedRequest | undefined; + } = { value: undefined }; + restore = installFetchStub(await readStreamFixture(), 200, captured); + const layer = makeResponsesModelLayer({ + model: "openrouter/jev", + apiKey: "sk-test", + retry: { baseDelayMs: 0, maxRetries: 0 }, + }); + const exit = await runPromiseExit( + gen(function* run() { + const model = yield* ResponsesModel; + return yield* model.generate([], { reasoningEffort: "auto" }); + }).pipe(provide(layer.pipe(layerProvide(FetchHttpClient.layer)))) + ); + assertSuccess(exit); + expect(captured.value).toBeDefined(); + expect(Object.hasOwn(captured.value?.body ?? {}, "reasoning")).toBe(false); + }); it("sends provider.only with fallbacks disabled on pinned runs", async () => { const captured: { value: CapturedRequest | undefined; diff --git a/src/providers/responses-model.ts b/src/providers/responses-model.ts index 81293e59..1972ca90 100644 --- a/src/providers/responses-model.ts +++ b/src/providers/responses-model.ts @@ -20,6 +20,7 @@ import { import type { Layer } from "effect/Layer"; import { effect, provide } from "effect/Layer"; +import { reasoningRequestFor } from "../harness/constants"; import type { ModelUsage } from "../harness/core"; import { ModelError } from "../harness/core"; import type { GenerateConfig } from "../harness/model"; @@ -186,10 +187,7 @@ export function generate( : undefined, }), ...definedValues({ - reasoning: - genConfig.reasoningEffort !== undefined - ? { effort: genConfig.reasoningEffort } - : undefined, + reasoning: reasoningRequestFor(genConfig.reasoningEffort), provider: sendProvider ? providerPreferences : undefined, }), ...definedValues({