Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions src/benchmarks/agent-cli/schema.ts
Original file line number Diff line number Diff line change
@@ -1,3 +1,5 @@
import type { ReasoningEffort } from "../../harness/constants";
import { ADAPTIVE_REASONING_EFFORT } from "../../harness/constants";
import type { ValueOf } from "../../internal/guards";

export const ORI_AGENTS = ["pi", "claude", "prime-agent", "omp"] as const;
Expand All @@ -22,6 +24,12 @@ export const ORI_REASONING_EFFORTS = [

export type OriReasoningEffort = ValueOf<typeof ORI_REASONING_EFFORTS>;

export function toOriReasoningEffort(
effort: ReasoningEffort
): OriReasoningEffort | undefined {
return effort === ADAPTIVE_REASONING_EFFORT ? undefined : effort;
}

export const AGENT_PACKAGE_PATTERN = /^[A-Za-z0-9@/:._^=+-]+$/;

export function isValidAgentPackage(value: string): boolean {
Expand Down
11 changes: 11 additions & 0 deletions src/benchmarks/benchmark-config.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,17 @@ describe("benchmark config", () => {
assertLeft(result);
});

it("accepts the explicit auto reasoning effort", () => {
const result = parseSchema(BenchmarkRunConfigSchema, {
benchmarkId: "gpqa_diamond",
model: "openrouter/jev",
reasoningEffort: "auto",
});

assertRight(result);
expect(result.right.reasoningEffort).toBe("auto");
});

it("parses injected benchmark configs with opaque options", () => {
const result = parseSchema(BenchmarkRunConfigSchema, {
benchmarkId: "injected_benchmark",
Expand Down
26 changes: 18 additions & 8 deletions src/benchmarks/deep-swe/solver.ts
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ import {
runAgentCli,
} from "../agent-cli/runner";
import type { HarborAgent } from "../agent-cli/schema";
import { isOriAgent } from "../agent-cli/schema";
import { isOriAgent, toOriReasoningEffort } from "../agent-cli/schema";
import type { InferenceOverride } from "../benchmark-config";
import type { AgentLoopInput } from "../harbor/agent-loop";
import { AGENT_ENV, probeSystemInfo, runAgentLoop } from "../harbor/agent-loop";
Expand Down Expand Up @@ -104,15 +104,25 @@ export function makeDeepSweSolver(
const task = loadTask(meta.taskId, tasksRoot);
const agent = opts.agent ?? "mini_swe";
const cliHarness = isOriAgent(agent) ? getOriHarness(agent) : undefined;
const baseCliOpts: AgentCliOpts =
const agentReasoningEffort = toOriReasoningEffort(
opts.inference.reasoningEffort
);
const baseCliOpts: AgentCliOpts | undefined =
opts.agentCli ??
definedValues({
model: opts.model,
apiKey: opts.apiKey,
endpointId: opts.endpointId,
sessionId: opts.sessionId,
agentReasoningEffort: opts.inference.reasoningEffort,
(agentReasoningEffort === undefined
? undefined
: definedValues({
model: opts.model,
apiKey: opts.apiKey,
endpointId: opts.endpointId,
sessionId: opts.sessionId,
agentReasoningEffort,
}));
if (baseCliOpts === undefined) {
return yield* new SolverError({
message: `deep-swe agent CLI requires a pinned reasoning effort (got "${opts.inference.reasoningEffort}")`,
});
}
const cliOpts: AgentCliOpts = {
...baseCliOpts,
appendSystemPrompt:
Expand Down
8 changes: 8 additions & 0 deletions src/benchmarks/search/core/request.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -196,6 +196,14 @@ describe("buildSearchRequestBody", () => {
expect(body.temperature).toBe(0);
expect(body.reasoning).toEqual({ effort: "high" });
});
it("omits reasoning on the wire when effort is auto", () => {
const body = buildSearchRequestBody({
...BASE,
lane: lane({}),
reasoningEffort: "auto",
});
expect(Object.hasOwn(body, "reasoning")).toBe(false);
});
it("threads the candidate models list and omits it by default", () => {
const routed = buildSearchRequestBody({
...BASE,
Expand Down
3 changes: 2 additions & 1 deletion src/benchmarks/search/core/request.ts
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@ import type {
} from "@openrouter/sdk/models";

import type { CostTier, ReasoningEffort } from "../../../harness/constants";
import { reasoningRequestFor } from "../../../harness/constants";
import type { ProviderSort } from "../../../internal/enums";
import { definedValues } from "../../../internal/guards";
import { buildAutoRouterPlugin } from "../../../providers/auto-router-plugin";
Expand Down Expand Up @@ -127,7 +128,7 @@ export function buildSearchRequestBody(
models: opts.models === undefined ? undefined : [...opts.models],
maxOutputTokens: opts.maxOutputTokens,
temperature: opts.temperature,
reasoning: { effort: opts.reasoningEffort },
reasoning: reasoningRequestFor(opts.reasoningEffort),
provider:
opts.sort !== undefined ||
opts.providerOrder !== undefined ||
Expand Down
24 changes: 17 additions & 7 deletions src/benchmarks/swe-atlas/solver.ts
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,7 @@ import {
runAgentCli,
} from "../agent-cli/runner";
import type { HarborAgent } from "../agent-cli/schema";
import { isOriAgent } from "../agent-cli/schema";
import { isOriAgent, toOriReasoningEffort } from "../agent-cli/schema";
import type { InferenceOverride } from "../benchmark-config";
import { AGENT_ENV, probeSystemInfo, runAgentLoop } from "../harbor/agent-loop";
import {
Expand Down Expand Up @@ -105,14 +105,24 @@ export function makeSweAtlasSolver(
const task = loadTask(meta.taskId, meta.track, tasksRoot);
const agent = opts.agent ?? "mini_swe";
const cliHarness = isOriAgent(agent) ? getOriHarness(agent) : undefined;
const baseCliOpts: AgentCliOpts =
const agentReasoningEffort = toOriReasoningEffort(
opts.inference.reasoningEffort
);
const baseCliOpts: AgentCliOpts | undefined =
opts.agentCli ??
definedValues({
model: opts.model,
apiKey: opts.apiKey,
endpointId: opts.endpointId,
agentReasoningEffort: opts.inference.reasoningEffort,
(agentReasoningEffort === undefined
? undefined
: definedValues({
model: opts.model,
apiKey: opts.apiKey,
endpointId: opts.endpointId,
agentReasoningEffort,
}));
if (baseCliOpts === undefined) {
return yield* new SolverError({
message: `swe-atlas agent CLI requires a pinned reasoning effort (got "${opts.inference.reasoningEffort}")`,
});
}
const cliOpts: AgentCliOpts = {
...baseCliOpts,
appendSystemPrompt: joinAgentPrompts(
Expand Down
54 changes: 54 additions & 0 deletions src/cli/index.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,33 @@ import { buildBenchmarkConfig, parseArgs } from ".";
describe("bench-harness CLI", () => {
it("defaults --reasoning-effort to high", () => {
expect(parseArgs([]).reasoningEffort).toBe("high");
expect(parseArgs(["--model", "openai/gpt-5"]).reasoningEffort).toBe("high");
});

it("defaults --reasoning-effort to auto for openrouter/jev", () => {
expect(parseArgs(["--model", "openrouter/jev"]).reasoningEffort).toBe(
"auto"
);
expect(parseArgs(["--model", "openrouter/jev:nitro"]).reasoningEffort).toBe(
"auto"
);
expect(
parseArgs(["--model", "openrouter/jev", "--reasoning-effort", "high"])
.reasoningEffort
).toBe("high");
expect(
parseArgs(["--model", "openrouter/jev", "--reasoning-effort", "auto"])
.reasoningEffort
).toBe("auto");
});

it("rejects --reasoning-effort auto for models without adaptive effort", () => {
expect(() =>
parseArgs(["--model", "openai/gpt-5", "--reasoning-effort", "auto"])
).toThrow("--reasoning-effort auto is only supported for");
expect(() => parseArgs(["--reasoning-effort", "auto"])).toThrow(
"--reasoning-effort auto is only supported for"
);
});

it("accepts an explicit --reasoning-effort", () => {
Expand Down Expand Up @@ -141,6 +168,33 @@ describe("bench-harness CLI", () => {
});
});

it("does not derive an ori agent reasoning effort from auto", () => {
expect(() =>
buildBenchmarkConfig({
benchmarkId: "terminal_bench",
model: "openrouter/jev",
panelConfig: undefined,
artifactDir: undefined,
endpointId: undefined,
imageDetail: undefined,
reasoningEffort: "auto",
})
).toThrow();
const config = buildBenchmarkConfig({
benchmarkId: "terminal_bench",
model: "openrouter/jev",
panelConfig: { agentReasoningEffort: "high" },
artifactDir: undefined,
endpointId: undefined,
imageDetail: undefined,
reasoningEffort: "auto",
});
expect(config).toMatchObject({
reasoningEffort: "auto",
agentReasoningEffort: "high",
});
});

it("preserves explicit ori agent reasoning effort", () => {
const config = buildBenchmarkConfig({
benchmarkId: "terminal_bench",
Expand Down
26 changes: 22 additions & 4 deletions src/cli/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -18,11 +18,14 @@ import { DracoPanelConfigSchema } from "../benchmarks/draco/schemas";
import { benchmarkIds, getBenchmark } from "../benchmarks/registry";
import type { CostTier, ReasoningEffort } from "../harness/constants";
import {
ADAPTIVE_REASONING_EFFORT,
ADAPTIVE_REASONING_EFFORT_MODELS,
COST_TIERS,
DEFAULT_REASONING_EFFORT,
defaultReasoningEffortFor,
ImageDetail,
IMAGE_DETAIL_VALUES,
REASONING_EFFORTS,
supportsAdaptiveReasoningEffort,
} from "../harness/constants";
import { makeProgressReporter } from "../harness/progress";
import { runHarnessPromise } from "../internal/effect-logger";
Expand Down Expand Up @@ -72,7 +75,10 @@ export function parseArgs(argv: readonly string[]): CliArgs {
resumeId: get("--resume-id"),
imageDetail: validateImageDetail(get("--image-detail")),
costTier: validateCostTier(get("--cost-tier")),
reasoningEffort: validateReasoningEffort(get("--reasoning-effort")),
reasoningEffort: validateReasoningEffort(
get("--reasoning-effort"),
get("--model")
),
};
}

Expand Down Expand Up @@ -304,15 +310,26 @@ function validateCostTier(raw: string | undefined): CostTier | undefined {
return raw;
}

function validateReasoningEffort(raw: string | undefined): ReasoningEffort {
function validateReasoningEffort(
raw: string | undefined,
model: string | undefined
): ReasoningEffort {
if (raw === undefined) {
return DEFAULT_REASONING_EFFORT;
return defaultReasoningEffortFor(model);
}
if (!isMember(raw, REASONING_EFFORTS)) {
throw new Error(
`--reasoning-effort must be one of: ${REASONING_EFFORTS.join(", ")} (got "${raw}")`
);
}
if (
raw === ADAPTIVE_REASONING_EFFORT &&
(model === undefined || !supportsAdaptiveReasoningEffort(model))
) {
throw new Error(
`--reasoning-effort ${ADAPTIVE_REASONING_EFFORT} is only supported for: ${ADAPTIVE_REASONING_EFFORT_MODELS.join(", ")} (got model "${model ?? ""}")`
);
}
return raw;
}

Expand Down Expand Up @@ -372,6 +389,7 @@ function buildSchemaValidatedConfig(opts: {
: undefined;
if (
optionsSchema !== undefined &&
reasoningEffort !== ADAPTIVE_REASONING_EFFORT &&
Object.hasOwn(optionsSchema.shape, "agentReasoningEffort") &&
!(
typeof panelConfig === "object" &&
Expand Down
18 changes: 18 additions & 0 deletions src/harness/constants.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
import { describe, expect, it } from "bun:test";

import { defaultReasoningEffortFor, reasoningRequestFor } from "./constants";

describe("reasoning effort constants", () => {
it("defaults openrouter/jev to auto and everything else to high", () => {
expect(defaultReasoningEffortFor("openrouter/jev")).toBe("auto");
expect(defaultReasoningEffortFor("openrouter/jev:nitro")).toBe("auto");
expect(defaultReasoningEffortFor("openai/gpt-5")).toBe("high");
expect(defaultReasoningEffortFor(undefined)).toBe("high");
});

it("maps auto to no reasoning object and pins every other effort", () => {
expect(reasoningRequestFor("auto")).toBeUndefined();
expect(reasoningRequestFor("high")).toEqual({ effort: "high" });
expect(reasoningRequestFor("none")).toEqual({ effort: "none" });
});
});
32 changes: 31 additions & 1 deletion src/harness/constants.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
import type { ValueOf } from "../internal/guards";

export const REASONING_EFFORTS = [
export const ADAPTIVE_REASONING_EFFORT = "auto";

export const PINNED_REASONING_EFFORTS = [
"xhigh",
"high",
"medium",
Expand All @@ -9,10 +11,38 @@ export const REASONING_EFFORTS = [
"none",
] as const;

export type PinnedReasoningEffort = ValueOf<typeof PINNED_REASONING_EFFORTS>;

export const REASONING_EFFORTS = [
...PINNED_REASONING_EFFORTS,
ADAPTIVE_REASONING_EFFORT,
] as const;

export type ReasoningEffort = ValueOf<typeof REASONING_EFFORTS>;

export const DEFAULT_REASONING_EFFORT: ReasoningEffort = "high";

export const ADAPTIVE_REASONING_EFFORT_MODELS = ["openrouter/jev"] as const;

export function supportsAdaptiveReasoningEffort(model: string): boolean {
const baseModel = model.split(":")[0];
return ADAPTIVE_REASONING_EFFORT_MODELS.some((slug) => slug === baseModel);
}

export function defaultReasoningEffortFor(
model: string | undefined
): ReasoningEffort {
return model !== undefined && supportsAdaptiveReasoningEffort(model)
? ADAPTIVE_REASONING_EFFORT
: DEFAULT_REASONING_EFFORT;
}

export function reasoningRequestFor(
effort: ReasoningEffort
): { readonly effort: PinnedReasoningEffort } | undefined {
return effort === ADAPTIVE_REASONING_EFFORT ? undefined : { effort };
}

export const COST_TIERS = ["low", "medium", "high", "xhigh", "max"] as const;

export type CostTier = ValueOf<typeof COST_TIERS>;
Expand Down
3 changes: 2 additions & 1 deletion src/judge/judge.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ import type { Effect } from "effect/Effect";
import { fail, flatMap, mapError, succeed } from "effect/Effect";

import type { ReasoningEffort } from "../harness/constants";
import { reasoningRequestFor } from "../harness/constants";
import type { ModelUsage } from "../harness/core";
import { ModelError } from "../harness/core";
import { Either } from "../internal/either";
Expand Down Expand Up @@ -67,7 +68,7 @@ export function judgeCall<T>(
text,
instructions: spec.instructions,
temperature: config.temperature,
reasoning: { effort: config.reasoningEffort },
reasoning: reasoningRequestFor(config.reasoningEffort),
});
const sendOptions = definedValues({
timeoutMs: config.timeoutMs ?? DEFAULT_JUDGE_TIMEOUT_MS,
Expand Down
20 changes: 20 additions & 0 deletions src/providers/responses-model.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -360,6 +360,26 @@ describe("responses-model", () => {
expect(captured.value?.body["provider"]).toBeUndefined();
expect(captured.value?.headers["x-or-endpoint-id"]).toBe("endpoint-1");
});
it("omits reasoning on the wire when effort is auto", async () => {
const captured: {
value: CapturedRequest | undefined;
} = { value: undefined };
restore = installFetchStub(await readStreamFixture(), 200, captured);
const layer = makeResponsesModelLayer({
model: "openrouter/jev",
apiKey: "sk-test",
retry: { baseDelayMs: 0, maxRetries: 0 },
});
const exit = await runPromiseExit(
gen(function* run() {
const model = yield* ResponsesModel;
return yield* model.generate([], { reasoningEffort: "auto" });
}).pipe(provide(layer.pipe(layerProvide(FetchHttpClient.layer))))
);
assertSuccess(exit);
expect(captured.value).toBeDefined();
expect(Object.hasOwn(captured.value?.body ?? {}, "reasoning")).toBe(false);
});
it("sends provider.only with fallbacks disabled on pinned runs", async () => {
const captured: {
value: CapturedRequest | undefined;
Expand Down
Loading
Loading