From 51d3866f69967a700c86c180d5629237778132a5 Mon Sep 17 00:00:00 2001 From: Oleg Zuev Date: Tue, 22 Sep 2026 13:41:54 +0700 Subject: [PATCH] Route reasoning effort and fast mode per turn with Jev Wirebot can now ask Jev, TypeSafe's decision model, to pick the reasoning effort for each user message instead of using one fixed level for every conversation. With WIREBOT_JEV_API_KEY set, the message text, whether it starts a new task, the attachment count and the connector name go to OpenRouter's Decisions endpoint before turn/start, and the chosen level (among the ones the selected model reports, from low up to ultra) is passed as the turn's effort. WIREBOT_JEV_FAST=auto additionally lets Jev decide per turn whether the fast service tier is worth it, using the serviceTierForTurn override so the choice never sticks to the thread. Routing is advisory and opt-in: a timeout, an API error or an unsupported level falls back to the configured settings, scheduled runs are untouched, and every decision is logged with its probabilities. The API key is scrubbed from Codex subprocess environments like the other bridge credentials. --- .env.example | 10 ++ CHANGELOG.md | 9 ++ README.md | 13 +++ src/channels/config-fields.ts | 14 +-- src/codex/config-service.ts | 13 +++ src/codex/service.ts | 59 ++++++++-- src/config/env.ts | 40 +++++++ src/index.ts | 20 ++++ src/routing/jev.ts | 101 +++++++++++++++++ src/routing/turn-router.ts | 198 ++++++++++++++++++++++++++++++++++ 10 files changed, 458 insertions(+), 19 deletions(-) create mode 100644 src/routing/jev.ts create mode 100644 src/routing/turn-router.ts diff --git a/.env.example b/.env.example index fa0670a..702af56 100644 --- a/.env.example +++ b/.env.example @@ -37,6 +37,16 @@ HOST=127.0.0.1 PORT=8787 LOG_LEVEL=info +# Optional: let Jev (TypeSafe's decision model, via OpenRouter) choose the +# reasoning effort for each user message. Set an OpenRouter API key to enable it; +# WIREBOT_JEV_FAST=auto additionally lets Jev pick Codex's fast service tier for +# quick replies (fast mode costs more credits). Message text is sent to OpenRouter. +# WIREBOT_JEV_API_KEY=sk-or-v1-replace-me +# WIREBOT_JEV_EFFORT=auto +# WIREBOT_JEV_FAST=off +# WIREBOT_JEV_FAST_TIER=fast +# WIREBOT_JEV_MODEL=~typesafe/jev-latest + # Wirebot pins its own Codex CLI version, so the pinned Codex should not # self-check for updates. CODEX_CHECK_UPDATES=false diff --git a/CHANGELOG.md b/CHANGELOG.md index 8e471f1..2082339 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,15 @@ All notable changes to Wirebot are documented in this file. ## [Unreleased] +### Added + +- Optional per-message reasoning-effort routing with [Jev](https://openrouter.ai/~typesafe/jev-latest), + TypeSafe's decision model, through OpenRouter. With `WIREBOT_JEV_API_KEY` set, each user turn + gets the effort level Jev picks among the ones the selected model supports (`low` for a + greeting, `ultra` for a large research task, and so on), falling back to the configured effort + on any error or timeout. `WIREBOT_JEV_FAST=auto` also lets Jev choose Codex's fast service tier + for quick replies on a per-turn basis. + ## [0.3.3] - 2026-09-21 ### Changed diff --git a/README.md b/README.md index d809767..bb5ce6c 100644 --- a/README.md +++ b/README.md @@ -146,6 +146,11 @@ and reverse-proxy that origin to the container's port 8787 (publish it in your c | `CODEX_API_KEY` | unset | OpenAI API key for headless Codex auth | | `HOST` / `PORT` | `0.0.0.0` / `8787` | Mini App listener (container default) | | `LOG_LEVEL` | `info` | `debug`, `info`, `warn`, or `error` | +| `WIREBOT_JEV_API_KEY` | unset | OpenRouter key; enables Jev turn routing | +| `WIREBOT_JEV_EFFORT` | `auto` | `off` keeps the configured effort | +| `WIREBOT_JEV_FAST` | `off` | `auto` lets Jev pick the fast tier | +| `WIREBOT_JEV_FAST_TIER` | `fast` | Service tier id used for fast replies | +| `WIREBOT_JEV_MODEL` | `~typesafe/jev-latest` | OpenRouter model id for the router | ## Authenticate Codex @@ -199,6 +204,14 @@ Wirebot connects through the Discord Gateway with [discord.js](https://github.co The connector is deliberately text-only. It does not download Discord attachments, stickers, or polls, and it never uploads generated files; both inbound and outbound omissions are stated in the conversation. Model-authored mentions and link previews are suppressed on every outbound create and edit. Set `DISCORD_BOT_TOKEN` and `DISCORD_ALLOWED_USER_IDS` together to enable it, then follow [docs/discord.md](docs/discord.md) for the Developer Portal intent, invite permissions, user IDs, and first run. +## Automatic reasoning effort with Jev + +Reasoning effort and speed are normally fixed in Codex settings for everyone who uses the bot. Wirebot can instead ask [Jev](https://openrouter.ai/~typesafe/jev-latest), TypeSafe's decision model, to pick them per message. Jev does not generate text: it answers typed questions about the message with calibrated probabilities in well under a second, so a greeting gets `low`, a typo fix `medium`, a debugging session `high`, a migration design `xhigh`, and a large research task `ultra` — whatever levels the selected Codex model reports. + +Set `WIREBOT_JEV_API_KEY` to an [OpenRouter](https://openrouter.ai/) API key to enable it. Before each user turn Wirebot sends the message text, whether it starts a new task, the attachment count, and the connector name to OpenRouter's Decisions endpoint, then passes the chosen effort to `turn/start`. Scheduled runs keep their own effort settings. Routing is advisory: a timeout (2.5 s), an API error, or a level the model does not support falls back to the configured effort, and the decision is logged with its probabilities at `info` level. + +`WIREBOT_JEV_FAST=auto` additionally lets Jev decide per turn whether a fast reply is worth it: short conversational messages and quick lookups run on the model's fast service tier (`WIREBOT_JEV_FAST_TIER`, `fast` by default, matching Codex's Fast mode), everything else on the standard tier. Fast mode consumes credits at a higher rate, so this stays off unless enabled. Message text leaves the host for this feature; do not enable it where that is not acceptable. + ## Scheduled runs Ask Codex naturally, for example, “Every weekday at 9, check this project for failed CI runs” or “Revisit this task every hour and notify me only if something changed.” Wirebot exposes a host-managed `automation_update` tool to new Codex tasks and stores each schedule with an explicit time zone. A task created before upgrading does not have that tool in its persisted definition; send `/new` once before asking it to create or edit schedules. `/schedules` remains available for viewing them. diff --git a/src/channels/config-fields.ts b/src/channels/config-fields.ts index a6ed276..87d4a1b 100644 --- a/src/channels/config-fields.ts +++ b/src/channels/config-fields.ts @@ -1,7 +1,7 @@ import { ConfigValidationError, type EditableConfigSnapshot, - type ModelCapability, + selectedModel, } from "../codex/config-service.js"; import { errorMessage } from "../shared/errors.js"; @@ -58,7 +58,7 @@ export function fieldOptions( snapshot: EditableConfigSnapshot, field: ConfigFieldKey, ): FieldOption[] { - const model = currentModel(snapshot); + const model = selectedModel(snapshot); switch (field) { case "model": return snapshot.capabilities.models.map((candidate) => ({ @@ -191,13 +191,3 @@ export async function applyConfigValue( : `⚠️ ${errorMessage(error)}`; } } - -function currentModel(snapshot: EditableConfigSnapshot): ModelCapability | undefined { - const selected = snapshot.values.model; - const models = snapshot.capabilities.models; - if (selected !== null) { - const match = models.find((model) => model.model === selected); - if (match !== undefined) return match; - } - return models.find((model) => model.isDefault) ?? models[0]; -} diff --git a/src/codex/config-service.ts b/src/codex/config-service.ts index b1c763c..fbc48bf 100644 --- a/src/codex/config-service.ts +++ b/src/codex/config-service.ts @@ -170,6 +170,19 @@ export interface EditableConfigSnapshot { readonly validation: ConfigValidationResult; } +/** The model the config selects, or the catalog default when it names none. */ +export function selectedModel( + snapshot: Pick, +): ModelCapability | undefined { + const selected = snapshot.values.model; + const models = snapshot.capabilities.models; + if (selected !== null) { + const match = models.find((model) => model.model === selected); + if (match !== undefined) return match; + } + return models.find((model) => model.isDefault) ?? models[0]; +} + export class ConfigValidationError extends BridgeError { public readonly issues: readonly ConfigValidationIssue[]; diff --git a/src/codex/service.ts b/src/codex/service.ts index 19d3b57..723a93c 100644 --- a/src/codex/service.ts +++ b/src/codex/service.ts @@ -135,6 +135,7 @@ type CodexTurnSettings = Readonly< | "sandboxPolicy" | "model" | "serviceTier" + | "serviceTierForTurn" | "effort" | "summary" | "personality" @@ -154,9 +155,32 @@ type ExplicitSkillInputProvider = ( text: string, ) => readonly ExplicitSkillInput[] | Promise; +/** What a turn router learns about a user message before the turn starts. */ +export interface TurnRoutingRequest { + readonly conversationKey: string; + readonly connector: string; + /** Message text after voice transcription. */ + readonly text: string; + /** True when this message starts a new Codex thread. */ + readonly newThread: boolean; + readonly attachmentCount: number; +} + +/** Per-turn overrides layered over the configured turn settings. */ +export interface TurnRoutingDecision { + readonly effort?: string; + readonly serviceTierForTurn?: string; +} + +type TurnRoutingProvider = ( + request: TurnRoutingRequest, +) => TurnRoutingDecision | Promise; + export interface CodexServiceProviders { readonly effectiveSettings?: EffectiveCodexSettingsProvider; readonly explicitSkillInputs?: ExplicitSkillInputProvider; + /** Optional per-turn settings routing for user turns (not scheduled runs). */ + readonly turnRouting?: TurnRoutingProvider; /** Static deployment-environment context added to every turn. */ readonly environmentContext?: ApplicationContext; readonly externalAuthTokens?: ( @@ -182,6 +206,7 @@ export class CodexService { readonly #remoteClientContextEnabled: () => boolean; readonly #effectiveSettings: EffectiveCodexSettingsProvider; readonly #explicitSkillInputs: ExplicitSkillInputProvider; + readonly #turnRouting: TurnRoutingProvider | undefined; readonly #environmentContext: ApplicationContext | undefined; readonly #externalAuthTokens: CodexServiceProviders["externalAuthTokens"]; #pauseGate: Deferred | undefined; @@ -215,6 +240,7 @@ export class CodexService { this.#remoteClientContextEnabled = remoteClientContextEnabled; this.#effectiveSettings = providers.effectiveSettings ?? (() => ({})); this.#explicitSkillInputs = providers.explicitSkillInputs ?? (() => []); + this.#turnRouting = providers.turnRouting; this.#environmentContext = providers.environmentContext; this.#externalAuthTokens = providers.externalAuthTokens; rpc.onNotification((notification) => this.handleNotification(notification)); @@ -338,12 +364,17 @@ export class CodexService { this.#effectiveSettings(), this.#explicitSkillInputs(prepared), ]); - const threadId = await this.ensureThread( - conversationKey, - connector, - ephemeral, - settings.thread ?? {}, - ); + const newThread = ephemeral || this.#conversations.get(conversationKey) === undefined; + const [threadId, routing] = await Promise.all([ + this.ensureThread(conversationKey, connector, ephemeral, settings.thread ?? {}), + this.routeTurn({ + conversationKey, + connector, + text: prepared, + newThread, + attachmentCount: attachments.length, + }), + ]); const session = this.requireSession(threadId); this.#conversationSessions.set(conversationKey, session); session.adoptPresenter(conversationKey, connector, responder, invocation); @@ -356,7 +387,7 @@ export class CodexService { conversationKey, connector, input: [...createTurnInput(prepared, connector, attachments), ...skillInputs], - turnSettings: settings.turn ?? {}, + turnSettings: { ...(settings.turn ?? {}), ...routing }, onStarted: () => { started = true; }, @@ -388,6 +419,20 @@ export class CodexService { } } + /** Routing is advisory: a failing router never fails the turn. */ + private async routeTurn(request: TurnRoutingRequest): Promise { + if (this.#turnRouting === undefined) return {}; + try { + return await this.#turnRouting(request); + } catch (error) { + this.#logger.warn("Turn routing failed; using the configured settings", { + conversationKey: request.conversationKey, + error: errorMessage(error), + }); + return {}; + } + } + public async runScheduledTurn(request: ScheduledTurnRequest): Promise { await this.enterJob(); try { diff --git a/src/config/env.ts b/src/config/env.ts index c4f2e7a..5a0b150 100644 --- a/src/config/env.ts +++ b/src/config/env.ts @@ -31,6 +31,12 @@ const envSchema = z.object({ HOST: z.string().min(1).default("127.0.0.1"), PORT: z.coerce.number().int().min(1).max(65_535).default(8787), LOG_LEVEL: z.enum(["debug", "info", "warn", "error"]).default("info"), + WIREBOT_JEV_API_KEY: z.string().min(1).optional(), + WIREBOT_JEV_MODEL: z.string().min(1).default("~typesafe/jev-latest"), + WIREBOT_JEV_ENDPOINT: z.url().default("https://openrouter.ai/api/alpha/decisions"), + WIREBOT_JEV_EFFORT: z.enum(["auto", "off"]).default("auto"), + WIREBOT_JEV_FAST: z.enum(["auto", "off"]).default("off"), + WIREBOT_JEV_FAST_TIER: z.string().min(1).default("fast"), }); /** @@ -55,6 +61,7 @@ export const bridgeOnlyEnvironmentKeys: ReadonlySet | undefined; } +/** Jev turn routing through OpenRouter; present only when an API key is set. */ +export interface JevConfig { + readonly apiKey: string; + readonly model: string; + readonly endpoint: string; + /** Let Jev choose the reasoning effort for each user turn. */ + readonly effort: boolean; + /** Let Jev choose per turn whether to use the fast service tier. */ + readonly fast: boolean; + readonly fastTier: string; +} + export interface AppConfig { readonly telegram: TelegramConfig | undefined; readonly telegramApiBase: string; @@ -103,6 +122,7 @@ export interface AppConfig { readonly host: string; readonly port: number; readonly logLevel: LogLevel; + readonly jev: JevConfig | undefined; } export function loadAppConfig(environment: NodeJS.ProcessEnv = process.env): AppConfig { @@ -142,6 +162,26 @@ export function loadAppConfig(environment: NodeJS.ProcessEnv = process.env): App host: parsed.HOST, port: parsed.PORT, logLevel: parsed.LOG_LEVEL, + jev: jevConfigFromParsed(parsed), + }; +} + +function jevConfigFromParsed(parsed: z.infer): JevConfig | undefined { + if (parsed.WIREBOT_JEV_API_KEY === undefined) return undefined; + const effort = parsed.WIREBOT_JEV_EFFORT === "auto"; + const fast = parsed.WIREBOT_JEV_FAST === "auto"; + if (!effort && !fast) { + throw new Error( + "WIREBOT_JEV_API_KEY is set but both WIREBOT_JEV_EFFORT and WIREBOT_JEV_FAST are off; unset the key or enable one of them", + ); + } + return { + apiKey: parsed.WIREBOT_JEV_API_KEY, + model: parsed.WIREBOT_JEV_MODEL, + endpoint: parsed.WIREBOT_JEV_ENDPOINT, + effort, + fast, + fastTier: parsed.WIREBOT_JEV_FAST_TIER, }; } diff --git a/src/index.ts b/src/index.ts index 66690f6..0b44a7d 100644 --- a/src/index.ts +++ b/src/index.ts @@ -18,6 +18,8 @@ import { WirebotMcpServer } from "./mcp/server.js"; import { BrowserAuth } from "./miniapp/browser-auth.js"; import { MiniAppServer } from "./miniapp/server.js"; import { QuickTunnel } from "./miniapp/tunnel.js"; +import { OpenRouterJevClient } from "./routing/jev.js"; +import { JevTurnRouter } from "./routing/turn-router.js"; import { deferred } from "./shared/async.js"; import { errorMessage } from "./shared/errors.js"; import { atomicWriteFile, ensureDirectory, readFileIfExists } from "./shared/fs.js"; @@ -122,6 +124,7 @@ export async function runWirebot(): Promise { chatgptAuth === undefined ? undefined : () => Promise.resolve(chatgptAuth), ); let liveRuntime: CodexRuntimeService | undefined; + let turnRouter: JevTurnRouter | undefined; codex = new CodexService( rpc, conversations, @@ -134,6 +137,7 @@ export async function runWirebot(): Promise { { effectiveSettings: () => liveRuntime?.settings() ?? {}, explicitSkillInputs: (text) => liveRuntime?.skillInputs(text) ?? [], + turnRouting: (request) => turnRouter?.route(request) ?? {}, ...(config.container ? { environmentContext: createContainerEnvironmentContext() } : {}), ...(chatgptAuth === undefined ? {} @@ -160,6 +164,22 @@ export async function runWirebot(): Promise { logger.info("Codex is authenticated with the API key from CODEX_API_KEY"); } const configService = new CodexConfigService(rpc, config.workspace); + if (config.jev !== undefined) { + turnRouter = new JevTurnRouter({ + client: new OpenRouterJevClient(config.jev), + config: configService, + logger: logger.child({ component: "jev-router" }), + effort: config.jev.effort, + fast: config.jev.fast, + fastTier: config.jev.fastTier, + }); + logger.info("Jev turn routing is enabled", { + model: config.jev.model, + effort: config.jev.effort, + fast: config.jev.fast, + fastTier: config.jev.fastTier, + }); + } const runtime = new CodexRuntimeService({ rpc, codex, diff --git a/src/routing/jev.ts b/src/routing/jev.ts new file mode 100644 index 0000000..029ec31 --- /dev/null +++ b/src/routing/jev.ts @@ -0,0 +1,101 @@ +import { z } from "zod"; + +/** + * Jev is TypeSafe's decision model: it answers typed questions about a state + * with calibrated probabilities instead of generating text. Wirebot only uses + * the `choice` and `noul` (yes/no) question kinds. + */ +export type JevQuestion = + | { + readonly type: "choice"; + readonly instructions: string; + /** Option id → what that option means. */ + readonly criteria: Readonly>; + } + | { + readonly type: "noul"; + readonly instructions: string; + readonly criteria?: { readonly true: string; readonly false: string }; + }; + +export type JevState = Readonly>; + +export interface JevDecisionRequest { + readonly state: JevState; + readonly questions: Readonly>; +} + +const choiceAnswerSchema = z.object({ + type: z.literal("choice"), + choice: z.string(), + probabilities: z.record(z.string(), z.number()), + confidence: z.number(), +}); + +const noulAnswerSchema = z.object({ + type: z.literal("noul"), + /** Probability that the statement holds, 0–1. */ + noul: z.number(), +}); + +const decisionSchema = z.object({ + model: z.string().optional(), + answers: z.record( + z.string(), + z.discriminatedUnion("type", [choiceAnswerSchema, noulAnswerSchema]), + ), + usage: z.object({ input_tokens: z.number().optional(), cost: z.number().optional() }).optional(), +}); + +export type JevChoiceAnswer = z.infer; +export type JevNoulAnswer = z.infer; +export type JevDecision = z.infer; + +export interface JevDecisionClient { + decide(request: JevDecisionRequest, signal?: AbortSignal): Promise; +} + +export interface OpenRouterJevClientOptions { + /** OpenRouter Decisions endpoint. */ + readonly endpoint: string; + readonly apiKey: string; + /** OpenRouter model id, normally `~typesafe/jev-latest`. */ + readonly model: string; + readonly fetch?: typeof fetch; +} + +/** Calls Jev through OpenRouter's Decisions API. */ +export class OpenRouterJevClient implements JevDecisionClient { + readonly #options: OpenRouterJevClientOptions; + readonly #fetch: typeof fetch; + + public constructor(options: OpenRouterJevClientOptions) { + this.#options = options; + this.#fetch = options.fetch ?? globalThis.fetch; + } + + public async decide(request: JevDecisionRequest, signal?: AbortSignal): Promise { + const response = await this.#fetch(this.#options.endpoint, { + method: "POST", + headers: { + authorization: `Bearer ${this.#options.apiKey}`, + "content-type": "application/json", + "http-referer": "https://github.com/sadfun/wirebot", + "x-title": "Wirebot", + }, + body: JSON.stringify({ + model: this.#options.model, + state: request.state, + questions: request.questions, + }), + ...(signal === undefined ? {} : { signal }), + }); + if (!response.ok) { + const body = (await response.text().catch(() => "")).slice(0, 300); + throw new Error( + `Jev request failed with HTTP ${response.status}${body === "" ? "" : `: ${body}`}`, + ); + } + return decisionSchema.parse(await response.json()); + } +} diff --git a/src/routing/turn-router.ts b/src/routing/turn-router.ts new file mode 100644 index 0000000..0459957 --- /dev/null +++ b/src/routing/turn-router.ts @@ -0,0 +1,198 @@ +import { + type EditableConfigSnapshot, + type ModelCapability, + selectedModel, +} from "../codex/config-service.js"; +import type { TurnRoutingDecision, TurnRoutingRequest } from "../codex/service.js"; +import { errorMessage } from "../shared/errors.js"; +import type { Logger } from "../shared/logger.js"; +import { truncate } from "../shared/text.js"; +import type { JevDecision, JevDecisionClient, JevQuestion } from "./jev.js"; + +export interface TurnRouterConfigAccess { + read(): Promise>; +} + +export interface JevTurnRouterOptions { + readonly client: JevDecisionClient; + readonly config: TurnRouterConfigAccess; + readonly logger: Logger; + /** Let Jev pick the reasoning effort among the model's supported levels. */ + readonly effort: boolean; + /** Let Jev decide per turn whether the fast service tier is worth it. */ + readonly fast: boolean; + /** Service tier id used when Jev votes for a fast reply. */ + readonly fastTier: string; + readonly timeoutMs?: number; + /** Minimum "fast is better" probability that selects the fast tier. */ + readonly fastThreshold?: number; +} + +const defaultTimeoutMs = 2_500; +const defaultFastThreshold = 0.7; +const maxMessageChars = 8_000; + +const effortInstructions = + "The message was sent to an autonomous coding assistant that can read and edit files, run commands, and browse the web. Pick how much reasoning effort the assistant needs to handle it well."; + +/** + * Routing-oriented descriptions of Codex's effort levels. Levels the model + * reports but this table lacks fall back to Codex's own description. + */ +const effortCriteria: Readonly> = { + none: "no reasoning needed: a greeting, thanks, or a one-word acknowledgement", + minimal: + "a greeting, thanks, a one-word confirmation, or a trivial question answerable in one line", + low: "a greeting, thanks, a short confirmation, or a trivial question answerable in a line or two", + medium: + "an ordinary task: a small edit, a lookup with a short explanation, a summary, or a routine report", + high: "multi-step work: debugging, investigating logs or code across several files, or a task with several moving parts", + xhigh: + "a very hard or high-stakes single problem: architecture or migration design, ambiguous requirements, or risky changes needing deep analysis", + ultra: + "large-scale research or investigation: surveying many sources, repositories, or systems and synthesizing them into a report, plan, or comparison; work that splits into parallel sub-tasks", + max: "one unusually hard problem that needs the deepest possible reasoning, such as a subtle correctness proof or a root cause nobody could find", +}; + +const fastQuestion: JevQuestion = { + type: "noul", + instructions: + "Would the sender be better served by a quick, low-latency reply than by a slower, more thorough one?", + criteria: { + true: "a short conversational message, quick confirmation, or simple lookup where waiting would be annoying", + false: "substantial work where quality and completeness matter more than response speed", + }, +}; + +/** + * Picks per-turn Codex settings with a Jev decision before each user turn. + * Any failure falls back to the configured settings; routing never blocks a + * turn for longer than the timeout. + */ +export class JevTurnRouter { + readonly #client: JevDecisionClient; + readonly #config: TurnRouterConfigAccess; + readonly #logger: Logger; + readonly #effort: boolean; + readonly #fast: boolean; + readonly #fastTier: string; + readonly #timeoutMs: number; + readonly #fastThreshold: number; + + public constructor(options: JevTurnRouterOptions) { + this.#client = options.client; + this.#config = options.config; + this.#logger = options.logger; + this.#effort = options.effort; + this.#fast = options.fast; + this.#fastTier = options.fastTier; + this.#timeoutMs = options.timeoutMs ?? defaultTimeoutMs; + this.#fastThreshold = options.fastThreshold ?? defaultFastThreshold; + } + + public async route(request: TurnRoutingRequest): Promise { + const model = await this.currentModel(); + if (model === undefined) return {}; + const efforts = this.#effort + ? model.supportedReasoningEfforts.map((option) => option.reasoningEffort) + : []; + const askEffort = efforts.length > 1; + const askFast = this.#fast && this.fastTierAvailable(model); + if (!askEffort && !askFast) return {}; + + const questions: Record = {}; + if (askEffort) { + questions.effort = { + type: "choice", + instructions: effortInstructions, + criteria: Object.fromEntries( + model.supportedReasoningEfforts.map((option) => [ + option.reasoningEffort, + effortCriteria[option.reasoningEffort] ?? + (option.description === "" ? option.reasoningEffort : option.description), + ]), + ), + }; + } + if (askFast) questions.fast = fastQuestion; + + const startedAt = Date.now(); + let decision: JevDecision; + try { + decision = await this.#client.decide( + { + state: { + message: truncate(request.text, maxMessageChars), + conversation: request.newThread + ? "first message of a new task" + : "follow-up in an ongoing task", + attachments: request.attachmentCount, + channel: request.connector, + }, + questions, + }, + AbortSignal.timeout(this.#timeoutMs), + ); + } catch (error) { + this.#logger.warn("Jev turn routing failed; using the configured settings", { + conversationKey: request.conversationKey, + error: errorMessage(error), + }); + return {}; + } + + const result: { effort?: string; serviceTierForTurn?: string } = {}; + const effortAnswer = decision.answers.effort; + if (askEffort && effortAnswer?.type === "choice") { + if (efforts.includes(effortAnswer.choice)) result.effort = effortAnswer.choice; + else { + this.#logger.warn("Jev picked a reasoning effort the model does not support", { + conversationKey: request.conversationKey, + choice: effortAnswer.choice, + supported: efforts, + }); + } + } + const fastAnswer = decision.answers.fast; + if (askFast && fastAnswer?.type === "noul") { + result.serviceTierForTurn = + fastAnswer.noul >= this.#fastThreshold ? this.#fastTier : "default"; + } + this.#logger.info("Jev routed the turn", { + conversationKey: request.conversationKey, + newThread: request.newThread, + ...result, + ...(effortAnswer?.type === "choice" + ? { + effortConfidence: effortAnswer.confidence, + effortProbabilities: effortAnswer.probabilities, + } + : {}), + ...(fastAnswer?.type === "noul" ? { fastProbability: fastAnswer.noul } : {}), + latencyMs: Date.now() - startedAt, + ...(decision.model === undefined ? {} : { jevModel: decision.model }), + }); + return result; + } + + private async currentModel(): Promise { + try { + return selectedModel(await this.#config.read()); + } catch (error) { + this.#logger.warn("Jev turn routing skipped: the Codex model catalog is unavailable", { + error: errorMessage(error), + }); + return undefined; + } + } + + private fastTierAvailable(model: ModelCapability): boolean { + if (model.serviceTiers.some((tier) => tier.id === this.#fastTier)) return true; + this.#logger.debug("Jev fast routing skipped: the model has no such service tier", { + model: model.model, + fastTier: this.#fastTier, + serviceTiers: model.serviceTiers.map((tier) => tier.id), + }); + return false; + } +}