From e4f94a4ee48b3cd2cd4349c2b3cf23a6b68af742 Mon Sep 17 00:00:00 2001 From: Siarhei Yarkavy Date: Sat, 26 Sep 2026 13:04:09 +0300 Subject: [PATCH 1/8] fix: report 1050000-token window for GPT-6 Luna and Sol models codemie-opencode overrode models.dev limits with heuristic defaults that had no gpt-6 branch, showing ~128k instead of the vendor-published 1050000. Refs #581 --- .../__tests__/opencode-gpt55-routing.test.ts | 63 +++++++++++++++++ .../opencode/opencode-dynamic-models.ts | 6 +- .../opencode/opencode-model-configs.ts | 68 +++++++++++++++++++ .../plugins/pi/__tests__/pi.models.test.ts | 32 +++++++++ src/agents/plugins/pi/pi.models.ts | 4 ++ .../__tests__/vscode-models.test.ts | 13 +++- .../proxy/connectors/__tests__/vscode.test.ts | 26 ++++--- .../proxy/connectors/vscode-models.ts | 22 ++++++ 8 files changed, 221 insertions(+), 13 deletions(-) diff --git a/src/agents/plugins/__tests__/opencode-gpt55-routing.test.ts b/src/agents/plugins/__tests__/opencode-gpt55-routing.test.ts index e9a5c0ea2..b415963df 100644 --- a/src/agents/plugins/__tests__/opencode-gpt55-routing.test.ts +++ b/src/agents/plugins/__tests__/opencode-gpt55-routing.test.ts @@ -165,3 +165,66 @@ describe('GPT-5.6 → Responses API routing', () => { expect(OPENCODE_MODEL_CONFIGS['gpt-5.6-sol-2026-07-09']!.limit.context).toBe(1050000); }); }); + +describe('GPT-6 → Responses API routing', () => { + let convertApiModelToOpenCodeConfig: typeof import('../opencode/opencode-dynamic-models.js').convertApiModelToOpenCodeConfig; + let OPENCODE_MODEL_CONFIGS: typeof import('../opencode/opencode-model-configs.js').OPENCODE_MODEL_CONFIGS; + + beforeEach(async () => { + vi.resetModules(); + ({ convertApiModelToOpenCodeConfig } = await import('../opencode/opencode-dynamic-models.js')); + ({ OPENCODE_MODEL_CONFIGS } = await import('../opencode/opencode-model-configs.js')); + }); + + // ── Dynamic path (live catalogue) ────────────────────────────────────────── + + it('routes gpt-6-luna to Responses API via dynamic model conversion', () => { + const config = convertApiModelToOpenCodeConfig(makeLlmModel('gpt-6-luna')); + expect(config.use_responses_api).toBe(true); + }); + + it('routes openai.gpt-6-luna (vendor-prefixed) to Responses API via dynamic conversion', () => { + const config = convertApiModelToOpenCodeConfig(makeLlmModel('openai.gpt-6-luna')); + expect(config.use_responses_api).toBe(true); + }); + + it('routes gpt-6-sol-2026-09-22 (dated) to Responses API via dynamic conversion', () => { + const config = convertApiModelToOpenCodeConfig(makeLlmModel('gpt-6-sol-2026-09-22')); + expect(config.use_responses_api).toBe(true); + }); + + it('detects the gpt-6 family for vendor-prefixed ids', () => { + const config = convertApiModelToOpenCodeConfig(makeLlmModel('openai.gpt-6-sol')); + expect(config.family).toBe('gpt-6'); + }); + + it('dynamic gpt-6-luna reports context limit of 1050000', () => { + const config = convertApiModelToOpenCodeConfig(makeLlmModel('gpt-6-luna')); + expect(config.limit.context).toBe(1050000); + expect(config.limit.output).toBe(128000); + }); + + it('dynamic openai.gpt-6-luna reports context limit of 1050000', () => { + const config = convertApiModelToOpenCodeConfig(makeLlmModel('openai.gpt-6-luna')); + expect(config.limit.context).toBe(1050000); + }); + + // ── Static fallback path (OPENCODE_MODEL_CONFIGS) ────────────────────────── + + it('static config has gpt-6-luna with use_responses_api: true', () => { + expect(OPENCODE_MODEL_CONFIGS['gpt-6-luna']).toBeDefined(); + expect(OPENCODE_MODEL_CONFIGS['gpt-6-luna']!.use_responses_api).toBe(true); + }); + + it('static config gpt-6-sol supports tool_call', () => { + expect(OPENCODE_MODEL_CONFIGS['gpt-6-sol']!.tool_call).toBe(true); + }); + + it('static config gpt-6-luna reports context limit of 1050000', () => { + expect(OPENCODE_MODEL_CONFIGS['gpt-6-luna']!.limit.context).toBe(1050000); + }); + + it('static config gpt-6-sol reports context limit of 1050000', () => { + expect(OPENCODE_MODEL_CONFIGS['gpt-6-sol']!.limit.context).toBe(1050000); + }); +}); diff --git a/src/agents/plugins/opencode/opencode-dynamic-models.ts b/src/agents/plugins/opencode/opencode-dynamic-models.ts index 3ff431be7..209065422 100644 --- a/src/agents/plugins/opencode/opencode-dynamic-models.ts +++ b/src/agents/plugins/opencode/opencode-dynamic-models.ts @@ -28,7 +28,8 @@ import { logger } from '../../../utils/logger.js'; // // Naming conventions observed in CodeMie deployments: // Responses API → gpt-5-2-*, gpt-5.2-*, gpt-5.x-codex-*, gpt-5-x-codex-*, -// gpt-5.4-*, gpt-5-4-*, gpt-5.5-*, gpt-5-5-*, gpt-5.6-*, gpt-5-6-* +// gpt-5.4-*, gpt-5-4-*, gpt-5.5-*, gpt-5-5-*, gpt-5.6-*, gpt-5-6-*, +// gpt-6-*, gpt-6.* (e.g. gpt-6-luna, openai.gpt-6-luna) // Chat Completions → gpt-4*, gpt-5--*, o1/o3/o4*, gemini-*, claude-*, … // // Update this list whenever new Responses-API-only models are deployed. @@ -45,6 +46,7 @@ const RESPONSES_API_MODEL_PATTERNS: RegExp[] = [ /^gpt-5\.5-/, // gpt-5.5-2026-04-24 — same Azure restriction as gpt-5.4 /^gpt-5-5-/, // hyphenated variant of gpt-5.5-* /gpt-5[.-]6/, // gpt-5.6-* (e.g. openai.gpt-5.6-luna) — same Azure restriction (tools + reasoning_effort) + /gpt-6[.-]/, // gpt-6-* (e.g. gpt-6-luna, openai.gpt-6-luna) — Responses API for built-in tools ]; function isResponsesApiModel(id: string): boolean { @@ -58,6 +60,7 @@ function detectFamily(id: string): string { if (id.startsWith('gemini')) return 'gemini-2'; if (id.startsWith('gpt-4')) return 'gpt-4'; if (id.startsWith('gpt-5') || /gpt-5[.-]6/.test(id)) return 'gpt-5'; + if (id.startsWith('gpt-6') || /gpt-6[.-]/.test(id)) return 'gpt-6'; if (/^o[134]-/.test(id) || id === 'o1') return 'openai-reasoning'; if (id.startsWith('qwen')) return 'qwen3'; if (id.startsWith('deepseek')) return 'deepseek'; @@ -77,6 +80,7 @@ function detectLimits(id: string, family: string): { context: number; output: nu if (id.startsWith('gpt-4o')) return { context: 128000, output: 16384 }; if (id.startsWith('gpt-5.5') || id.startsWith('gpt-5-5')) return { context: 1050000, output: 128000 }; // Azure-published window for gpt-5.5 if (/gpt-5[.-]6/.test(id)) return { context: 1050000, output: 128000 }; // Azure-published window for gpt-5.6 + if (id.startsWith('gpt-6') || /gpt-6[.-]/.test(id)) return { context: 1050000, output: 128000 }; // Vendor-published window for gpt-6 (1,050,000 context, 128k output) if (id.startsWith('gpt-5')) return { context: 400000, output: 128000 }; if (/^o[134]-/.test(id) || id === 'o1') return { context: 200000, output: 100000 }; if (id.startsWith('qwen') || id.startsWith('moonshotai') || id.startsWith('kimi')) return { context: 262144, output: 131072 }; diff --git a/src/agents/plugins/opencode/opencode-model-configs.ts b/src/agents/plugins/opencode/opencode-model-configs.ts index 81ff5e0d8..e87ac3418 100644 --- a/src/agents/plugins/opencode/opencode-model-configs.ts +++ b/src/agents/plugins/opencode/opencode-model-configs.ts @@ -288,6 +288,66 @@ export const OPENCODE_MODEL_CONFIGS: Record = { } }, + // ── GPT-6 Models (vendor-published: 1,050,000 context, 128k output, Responses API) ── + 'gpt-6-luna': { + id: 'gpt-6-luna', + name: 'GPT-6 Luna', + displayName: 'GPT-6 Luna', + family: 'gpt-6', + tool_call: true, + reasoning: true, + attachment: true, + temperature: false, + structured_output: true, + use_responses_api: true, + modalities: { + input: ['text', 'image'], + output: ['text'] + }, + knowledge: '2026-05-18', + release_date: '2026-09-22', + last_updated: '2026-09-22', + open_weights: false, + cost: { + input: 0.10, + output: 0.50, + cache_read: 0.01 + }, + limit: { + context: 1050000, + output: 128000 + } + }, + 'gpt-6-sol': { + id: 'gpt-6-sol', + name: 'GPT-6 Sol', + displayName: 'GPT-6 Sol', + family: 'gpt-6', + tool_call: true, + reasoning: true, + attachment: true, + temperature: false, + structured_output: true, + use_responses_api: true, + modalities: { + input: ['text', 'image'], + output: ['text'] + }, + knowledge: '2026-05-18', + release_date: '2026-09-22', + last_updated: '2026-09-22', + open_weights: false, + cost: { + input: 2, + output: 10, + cache_read: 0.20 + }, + limit: { + context: 1050000, + output: 128000 + } + }, + // ── Claude Models ────────────────────────────────────────────────── 'claude-4-5-sonnet': { id: 'claude-4-5-sonnet', @@ -647,6 +707,14 @@ const MODEL_FAMILY_DEFAULTS: Record> = { modalities: { input: ['text', 'image', 'audio', 'video'], output: ['text'] }, limit: { context: 1048576, output: 65536 } }, + 'gpt-6': { + family: 'gpt-6', + reasoning: true, + attachment: true, + temperature: false, + modalities: { input: ['text', 'image'], output: ['text'] }, + limit: { context: 1050000, output: 128000 } + }, 'gpt': { family: 'gpt-5', reasoning: true, diff --git a/src/agents/plugins/pi/__tests__/pi.models.test.ts b/src/agents/plugins/pi/__tests__/pi.models.test.ts index f734c6164..47829cc06 100644 --- a/src/agents/plugins/pi/__tests__/pi.models.test.ts +++ b/src/agents/plugins/pi/__tests__/pi.models.test.ts @@ -253,3 +253,35 @@ describe('convertLlmModelToPiEntry — cost', () => { expect(entry.compat).toEqual({ forceAdaptiveThinking: true }); }); }); + +describe('convertLlmModelToPiEntry — GPT-6 limits and routing', () => { + beforeEach(() => { + vi.mocked(lookupPrice).mockReset(); + vi.mocked(lookupPrice).mockReturnValue(null); + }); + + function gpt6Model(id: string): LlmModel { + return llmModel({ base_name: id, deployment_name: id, label: id }); + } + + it('should report a 1050000-token window for gpt-6-luna', () => { + const entry = convertLlmModelToPiEntry(gpt6Model('gpt-6-luna')); + + expect(entry.contextWindow).toBe(1050000); + expect(entry.maxTokens).toBe(128000); + }); + + it('should report a 1050000-token window for a vendor-prefixed id', () => { + const entry = convertLlmModelToPiEntry(gpt6Model('openai.gpt-6-sol')); + + expect(entry.contextWindow).toBe(1050000); + expect(entry.maxTokens).toBe(128000); + }); + + it('should route gpt-6-luna via openai-responses with reasoning', () => { + const entry = convertLlmModelToPiEntry(gpt6Model('gpt-6-luna')); + + expect(entry.api).toBe('openai-responses'); + expect(entry.reasoning).toBe(true); + }); +}); diff --git a/src/agents/plugins/pi/pi.models.ts b/src/agents/plugins/pi/pi.models.ts index 8348de219..3ad9fa91f 100644 --- a/src/agents/plugins/pi/pi.models.ts +++ b/src/agents/plugins/pi/pi.models.ts @@ -24,6 +24,7 @@ const RESPONSES_API_PATTERNS: RegExp[] = [ /^gpt-5\.5-/, /^gpt-5-5-/, /gpt-5[.-]6/, + /gpt-6[.-]/, ]; export function classifyPiModel(modelId: string): PiModelClassification { @@ -69,6 +70,7 @@ function detectLimits(id: string): { contextWindow: number; maxTokens: number } if (id.startsWith('gpt-4.1')) return { contextWindow: 1048576, maxTokens: 32768 }; if (/^gpt-5\.5-/.test(id) || /^gpt-5-5-/.test(id)) return { contextWindow: 1050000, maxTokens: 128000 }; if (/gpt-5[.-]6/.test(id)) return { contextWindow: 1050000, maxTokens: 128000 }; + if (id.startsWith('gpt-6') || /gpt-6[.-]/.test(id)) return { contextWindow: 1050000, maxTokens: 128000 }; if (id.startsWith('gpt-5')) return { contextWindow: 400000, maxTokens: 128000 }; if (/^o[134]-/.test(id) || id === 'o1') return { contextWindow: 200000, maxTokens: 100000 }; if (id.startsWith('qwen') || id.startsWith('moonshotai') || id.startsWith('kimi')) { @@ -96,6 +98,8 @@ function isReasoningModel(id: string): boolean { id.startsWith('gemini') || id.startsWith('gpt-5') || /gpt-5[.-]6/.test(id) || + id.startsWith('gpt-6') || + /gpt-6[.-]/.test(id) || /^o[134]-/.test(id) || id === 'o1' || id.startsWith('deepseek') || diff --git a/src/cli/commands/proxy/connectors/__tests__/vscode-models.test.ts b/src/cli/commands/proxy/connectors/__tests__/vscode-models.test.ts index 00a4dc7ca..b864b1197 100644 --- a/src/cli/commands/proxy/connectors/__tests__/vscode-models.test.ts +++ b/src/cli/commands/proxy/connectors/__tests__/vscode-models.test.ts @@ -24,8 +24,17 @@ describe('findVsCodeCapabilityEntry', () => { expect(findVsCodeCapabilityEntry('openai.gpt-5.6-luna')?.family).toBe('gpt-5.6-luna'); }); - it('returns undefined for a model with no capability-table family', () => { - expect(findVsCodeCapabilityEntry('gpt-6-sol')).toBeUndefined(); + it('returns the table entry for a GPT-6 model', () => { + expect(findVsCodeCapabilityEntry('gpt-6-sol')).toMatchObject({ + family: 'gpt-6-sol', + apiType: 'responses', + maxInputTokens: 922000, + maxOutputTokens: 128000, + }); + }); + + it('resolves a vendor-prefixed GPT-6 tenant id to its family entry', () => { + expect(findVsCodeCapabilityEntry('openai.gpt-6-luna')?.family).toBe('gpt-6-luna'); }); }); diff --git a/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts b/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts index 85d537b09..c7d5e94fb 100644 --- a/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts +++ b/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts @@ -25,6 +25,8 @@ const EXPECTED_MODEL_IDS = [ 'gpt-5.6-luna-2026-07-09', 'gpt-5.6-sol-2026-07-09', 'gpt-5.6-terra-2026-07-09', + 'gpt-6-luna', + 'gpt-6-sol', 'gemini-3-flash', 'gemini-3.1-pro', 'gemini-3.5-flash', @@ -170,7 +172,7 @@ describe('writeVsCodeLanguageModelsConfigAtPath', () => { } }); - it('renders stateless Responses reasoning capabilities for GPT-5.5 and GPT-5.6', async () => { + it('renders stateless Responses reasoning capabilities for GPT-5.5, GPT-5.6 and GPT-6', async () => { await writeVsCodeLanguageModelsConfigAtPath(configPath, 'http://127.0.0.1:4001', 'gw-key'); const providers = await readProviders(); @@ -180,6 +182,8 @@ describe('writeVsCodeLanguageModelsConfigAtPath', () => { ['gpt-5.6-luna-2026-07-09', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], ['gpt-5.6-sol-2026-07-09', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], ['gpt-5.6-terra-2026-07-09', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], + ['gpt-6-luna', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], + ['gpt-6-sol', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], ]); for (const [id, efforts] of expectedEfforts) { @@ -404,24 +408,26 @@ describe('writeVsCodeLanguageModelsConfigAtPath', () => { describe('full tenant catalog', () => { const GPT_6_SOL = { base_name: 'gpt-6-sol', label: 'GPT-6 Sol' }; - it('writes every enabled tenant model in catalog order, including unknown families', async () => { - mockCatalog([...EXPECTED_MODEL_IDS, GPT_6_SOL]); + it('writes every enabled tenant model in catalog order, including table-backed GPT-6 models', async () => { + mockCatalog(EXPECTED_MODEL_IDS); const result = await writeVsCodeLanguageModelsConfigAtPath(configPath, 'http://127.0.0.1:4001', 'gw-key'); const providers = await readProviders(); const models = providers[0].models as Array>; - expect(result.modelCount).toBe(EXPECTED_MODEL_IDS.length + 1); - expect(models.map(model => model.id)).toEqual([...EXPECTED_MODEL_IDS, 'gpt-6-sol']); + expect(result.modelCount).toBe(EXPECTED_MODEL_IDS.length); + expect(models.map(model => model.id)).toEqual(EXPECTED_MODEL_IDS); expect(models.find(model => model.id === 'gpt-6-sol')).toMatchObject({ - name: 'GPT-6 Sol', + name: 'gpt-6-sol', apiType: 'responses', zeroDataRetentionEnabled: true, url: 'http://127.0.0.1:4001/v1/responses', + maxInputTokens: 922000, + maxOutputTokens: 128000, }); }); - it('keeps catalog order when an unknown model comes first', async () => { + it('keeps catalog order when a GPT-6 model comes first', async () => { mockCatalog([GPT_6_SOL, 'claude-sonnet-5', 'gpt-4.1']); await writeVsCodeLanguageModelsConfigAtPath(configPath, 'http://127.0.0.1:4001', 'gw-key'); @@ -432,15 +438,15 @@ describe('writeVsCodeLanguageModelsConfigAtPath', () => { }); it('rebuilds the list from the live catalog on every write, picking up new models', async () => { - mockCatalog([...EXPECTED_MODEL_IDS, GPT_6_SOL]); + mockCatalog(EXPECTED_MODEL_IDS); await writeVsCodeLanguageModelsConfigAtPath(configPath, 'http://127.0.0.1:4001', 'gw-key'); - mockCatalog([...EXPECTED_MODEL_IDS, GPT_6_SOL, 'claude-sonnet-6']); + mockCatalog([...EXPECTED_MODEL_IDS, 'claude-sonnet-6']); const result = await writeVsCodeLanguageModelsConfigAtPath(configPath, 'http://127.0.0.1:4001', 'gw-key'); const providers = await readProviders(); const models = providers[0].models as Array>; - expect(result.modelCount).toBe(EXPECTED_MODEL_IDS.length + 2); + expect(result.modelCount).toBe(EXPECTED_MODEL_IDS.length + 1); expect(models.map(model => model.id)).toContain('claude-sonnet-6'); }); diff --git a/src/cli/commands/proxy/connectors/vscode-models.ts b/src/cli/commands/proxy/connectors/vscode-models.ts index 6c7c45020..485393d3e 100644 --- a/src/cli/commands/proxy/connectors/vscode-models.ts +++ b/src/cli/commands/proxy/connectors/vscode-models.ts @@ -169,6 +169,28 @@ export const VS_CODE_CAPABILITY_TABLE: readonly VsCodeCapabilityEntry[] = [ maxInputTokens: 922000, maxOutputTokens: 128000, }, + { + family: 'gpt-6-luna', + apiType: 'responses', + vision: true, + thinking: true, + zeroDataRetentionEnabled: true, + supportsReasoningEffort: GPT_5_6_EFFORTS, + reasoningEffortFormat: 'responses', + maxInputTokens: 922000, + maxOutputTokens: 128000, + }, + { + family: 'gpt-6-sol', + apiType: 'responses', + vision: true, + thinking: true, + zeroDataRetentionEnabled: true, + supportsReasoningEffort: GPT_5_6_EFFORTS, + reasoningEffortFormat: 'responses', + maxInputTokens: 922000, + maxOutputTokens: 128000, + }, { family: 'gemini-3-flash', apiType: 'chat-completions', From 8417673e3208444b32ced569fbdcdf071f324184 Mon Sep 17 00:00:00 2001 From: Siarhei Yarkavy Date: Mon, 28 Sep 2026 19:04:32 +0300 Subject: [PATCH 2/8] fix(proxy): use dedicated GPT-6 efforts for VS Code capabilities Address review on #582: gpt-6-luna/sol now use GPT_6_EFFORTS instead of GPT_5_6_EFFORTS. Same values, isolated for future divergence. --- src/cli/commands/proxy/connectors/vscode-models.ts | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/cli/commands/proxy/connectors/vscode-models.ts b/src/cli/commands/proxy/connectors/vscode-models.ts index 485393d3e..1348279f9 100644 --- a/src/cli/commands/proxy/connectors/vscode-models.ts +++ b/src/cli/commands/proxy/connectors/vscode-models.ts @@ -36,6 +36,7 @@ const GPT_5_EFFORTS = ['minimal', 'low', 'medium', 'high'] as const; const GPT_5_2_EFFORTS = ['none', 'low', 'medium', 'high'] as const; const GPT_5_XHIGH_EFFORTS = ['none', 'low', 'medium', 'high', 'xhigh'] as const; const GPT_5_6_EFFORTS = ['none', 'low', 'medium', 'high', 'xhigh', 'max'] as const; +const GPT_6_EFFORTS = ['none', 'low', 'medium', 'high', 'xhigh', 'max'] as const; const GEMINI_FLASH_EFFORTS = ['minimal', 'low', 'medium', 'high'] as const; const GEMINI_PRO_EFFORTS = ['low', 'medium', 'high'] as const; const CLAUDE_EFFORTS = ['low', 'medium', 'high'] as const; @@ -175,7 +176,7 @@ export const VS_CODE_CAPABILITY_TABLE: readonly VsCodeCapabilityEntry[] = [ vision: true, thinking: true, zeroDataRetentionEnabled: true, - supportsReasoningEffort: GPT_5_6_EFFORTS, + supportsReasoningEffort: GPT_6_EFFORTS, reasoningEffortFormat: 'responses', maxInputTokens: 922000, maxOutputTokens: 128000, @@ -186,7 +187,7 @@ export const VS_CODE_CAPABILITY_TABLE: readonly VsCodeCapabilityEntry[] = [ vision: true, thinking: true, zeroDataRetentionEnabled: true, - supportsReasoningEffort: GPT_5_6_EFFORTS, + supportsReasoningEffort: GPT_6_EFFORTS, reasoningEffortFormat: 'responses', maxInputTokens: 922000, maxOutputTokens: 128000, From 5482355e9ef66d3f8c5bfafa7c0998f59e78606e Mon Sep 17 00:00:00 2001 From: Siarhei Yarkavy Date: Thu, 1 Oct 2026 12:59:34 +0300 Subject: [PATCH 3/8] test(proxy): cover chat and responses fallbacks for unknown models --- tests/integration/vscode-byok.test.ts | 22 ++++++++++++++++------ 1 file changed, 16 insertions(+), 6 deletions(-) diff --git a/tests/integration/vscode-byok.test.ts b/tests/integration/vscode-byok.test.ts index 765fe1d5e..4d0bf53d9 100644 --- a/tests/integration/vscode-byok.test.ts +++ b/tests/integration/vscode-byok.test.ts @@ -35,7 +35,12 @@ import { const GATEWAY_KEY = 'test-local-key'; const PROFILE_MODEL = 'profile-selected-model-that-must-not-be-used'; -const UNKNOWN_MODEL = 'gpt-6-sol'; +// Neutral unknown: no gpt-/claude-/codex prefix, so buildDefaultVsCodeCapability +// falls back to conservative chat-completions defaults. +const UNKNOWN_CHAT_MODEL = 'totally-unknown-model'; +// Responses unknown: gpt-7 major guarantees Responses defaults via +// isResponsesOnlyGpt, while the segment never matches a capability-table family. +const UNKNOWN_RESPONSES_MODEL = 'gpt-7-future-unknown-xyz'; interface StartedServer { server: Server; @@ -208,11 +213,12 @@ describe('VS Code BYOK model matrix', () => { // serve it directly from every family in the capability table so each // entry resolves as an exact match — the model matrix below then // exercises the request-forwarding behavior, not the resolver itself. - // One extra id with no capability family must still be written. + // Two extra ids with no capability family must still be written: + // a neutral chat-completions fallback and a Responses fallback. if (req.url?.startsWith('/v1/llm_models')) { res.setHeader('content-type', 'application/json'); res.end(JSON.stringify({ - data: [...VS_CODE_CAPABILITY_TABLE.map(entry => ({ id: entry.family })), { id: UNKNOWN_MODEL }], + data: [...VS_CODE_CAPABILITY_TABLE.map(entry => ({ id: entry.family })), { id: UNKNOWN_CHAT_MODEL }, { id: UNKNOWN_RESPONSES_MODEL }], })); return; } @@ -240,10 +246,14 @@ describe('VS Code BYOK model matrix', () => { provider => provider.name === 'CodeMie' && provider.vendor === 'customendpoint' ); - expect(codeMieProvider?.models).toHaveLength(VS_CODE_CAPABILITY_TABLE.length + 1); + expect(codeMieProvider?.models).toHaveLength(VS_CODE_CAPABILITY_TABLE.length + 2); expect(codeMieProvider?.models?.some(model => model.id === PROFILE_MODEL)).toBe(false); - expect(codeMieProvider?.models?.find(model => model.id === UNKNOWN_MODEL)).toMatchObject({ - id: UNKNOWN_MODEL, + expect(codeMieProvider?.models?.find(model => model.id === UNKNOWN_CHAT_MODEL)).toMatchObject({ + id: UNKNOWN_CHAT_MODEL, + apiType: 'chat-completions', + }); + expect(codeMieProvider?.models?.find(model => model.id === UNKNOWN_RESPONSES_MODEL)).toMatchObject({ + id: UNKNOWN_RESPONSES_MODEL, apiType: 'responses', zeroDataRetentionEnabled: true, }); From bb3fe16fb4a38b115aa949600befa1143a42829f Mon Sep 17 00:00:00 2001 From: Siarhei Yarkavy Date: Sun, 4 Oct 2026 21:20:53 +0300 Subject: [PATCH 4/8] fix(agents): match bare and future GPT-6 ids in OpenCode and Pi Replace startsWith('gpt-6') || /gpt-6[.-]/ with /gpt-6(?!\d)/ so bare vendor-prefixed ids (openai.gpt-6) and dotted variants (gpt-6.1-sol, future 6.2+) route via Responses API with 1050000/128000 limits, while gpt-60-style ids stay out. --- src/agents/plugins/opencode/opencode-dynamic-models.ts | 8 ++++---- src/agents/plugins/pi/pi.models.ts | 9 +++++---- 2 files changed, 9 insertions(+), 8 deletions(-) diff --git a/src/agents/plugins/opencode/opencode-dynamic-models.ts b/src/agents/plugins/opencode/opencode-dynamic-models.ts index 209065422..0b2b4aaa4 100644 --- a/src/agents/plugins/opencode/opencode-dynamic-models.ts +++ b/src/agents/plugins/opencode/opencode-dynamic-models.ts @@ -29,7 +29,7 @@ import { logger } from '../../../utils/logger.js'; // Naming conventions observed in CodeMie deployments: // Responses API → gpt-5-2-*, gpt-5.2-*, gpt-5.x-codex-*, gpt-5-x-codex-*, // gpt-5.4-*, gpt-5-4-*, gpt-5.5-*, gpt-5-5-*, gpt-5.6-*, gpt-5-6-*, -// gpt-6-*, gpt-6.* (e.g. gpt-6-luna, openai.gpt-6-luna) +// gpt-6-*, gpt-6.* (e.g. gpt-6-luna, gpt-6.1-sol, openai.gpt-6-luna, bare openai.gpt-6) // Chat Completions → gpt-4*, gpt-5--*, o1/o3/o4*, gemini-*, claude-*, … // // Update this list whenever new Responses-API-only models are deployed. @@ -46,7 +46,7 @@ const RESPONSES_API_MODEL_PATTERNS: RegExp[] = [ /^gpt-5\.5-/, // gpt-5.5-2026-04-24 — same Azure restriction as gpt-5.4 /^gpt-5-5-/, // hyphenated variant of gpt-5.5-* /gpt-5[.-]6/, // gpt-5.6-* (e.g. openai.gpt-5.6-luna) — same Azure restriction (tools + reasoning_effort) - /gpt-6[.-]/, // gpt-6-* (e.g. gpt-6-luna, openai.gpt-6-luna) — Responses API for built-in tools + /gpt-6(?!\d)/, // gpt-6-* / gpt-6.* (e.g. gpt-6-luna, gpt-6.1-sol, openai.gpt-6-luna, chatgpt-6.1-sol) — Responses API for built-in tools; (?!\d) keeps gpt-60-style ids out while staying ready for 6.2+ ]; function isResponsesApiModel(id: string): boolean { @@ -60,7 +60,7 @@ function detectFamily(id: string): string { if (id.startsWith('gemini')) return 'gemini-2'; if (id.startsWith('gpt-4')) return 'gpt-4'; if (id.startsWith('gpt-5') || /gpt-5[.-]6/.test(id)) return 'gpt-5'; - if (id.startsWith('gpt-6') || /gpt-6[.-]/.test(id)) return 'gpt-6'; + if (/gpt-6(?!\d)/.test(id)) return 'gpt-6'; if (/^o[134]-/.test(id) || id === 'o1') return 'openai-reasoning'; if (id.startsWith('qwen')) return 'qwen3'; if (id.startsWith('deepseek')) return 'deepseek'; @@ -80,7 +80,7 @@ function detectLimits(id: string, family: string): { context: number; output: nu if (id.startsWith('gpt-4o')) return { context: 128000, output: 16384 }; if (id.startsWith('gpt-5.5') || id.startsWith('gpt-5-5')) return { context: 1050000, output: 128000 }; // Azure-published window for gpt-5.5 if (/gpt-5[.-]6/.test(id)) return { context: 1050000, output: 128000 }; // Azure-published window for gpt-5.6 - if (id.startsWith('gpt-6') || /gpt-6[.-]/.test(id)) return { context: 1050000, output: 128000 }; // Vendor-published window for gpt-6 (1,050,000 context, 128k output) + if (/gpt-6(?!\d)/.test(id)) return { context: 1050000, output: 128000 }; // Vendor-published window for gpt-6 (1,050,000 context, 128k output; covers 6.1-sol, future 6.2+) if (id.startsWith('gpt-5')) return { context: 400000, output: 128000 }; if (/^o[134]-/.test(id) || id === 'o1') return { context: 200000, output: 100000 }; if (id.startsWith('qwen') || id.startsWith('moonshotai') || id.startsWith('kimi')) return { context: 262144, output: 131072 }; diff --git a/src/agents/plugins/pi/pi.models.ts b/src/agents/plugins/pi/pi.models.ts index 3ad9fa91f..9a4a47e43 100644 --- a/src/agents/plugins/pi/pi.models.ts +++ b/src/agents/plugins/pi/pi.models.ts @@ -24,7 +24,9 @@ const RESPONSES_API_PATTERNS: RegExp[] = [ /^gpt-5\.5-/, /^gpt-5-5-/, /gpt-5[.-]6/, - /gpt-6[.-]/, + // gpt-6-*/gpt-6.* (e.g. gpt-6-luna, gpt-6.1-sol, openai.gpt-6-luna, bare openai.gpt-6). + // (?!\d) keeps gpt-60-style ids out while staying ready for 6.2+. + /gpt-6(?!\d)/, ]; export function classifyPiModel(modelId: string): PiModelClassification { @@ -70,7 +72,7 @@ function detectLimits(id: string): { contextWindow: number; maxTokens: number } if (id.startsWith('gpt-4.1')) return { contextWindow: 1048576, maxTokens: 32768 }; if (/^gpt-5\.5-/.test(id) || /^gpt-5-5-/.test(id)) return { contextWindow: 1050000, maxTokens: 128000 }; if (/gpt-5[.-]6/.test(id)) return { contextWindow: 1050000, maxTokens: 128000 }; - if (id.startsWith('gpt-6') || /gpt-6[.-]/.test(id)) return { contextWindow: 1050000, maxTokens: 128000 }; + if (/gpt-6(?!\d)/.test(id)) return { contextWindow: 1050000, maxTokens: 128000 }; if (id.startsWith('gpt-5')) return { contextWindow: 400000, maxTokens: 128000 }; if (/^o[134]-/.test(id) || id === 'o1') return { contextWindow: 200000, maxTokens: 100000 }; if (id.startsWith('qwen') || id.startsWith('moonshotai') || id.startsWith('kimi')) { @@ -98,8 +100,7 @@ function isReasoningModel(id: string): boolean { id.startsWith('gemini') || id.startsWith('gpt-5') || /gpt-5[.-]6/.test(id) || - id.startsWith('gpt-6') || - /gpt-6[.-]/.test(id) || + /gpt-6(?!\d)/.test(id) || /^o[134]-/.test(id) || id === 'o1' || id.startsWith('deepseek') || From ef3bca676a363c4b3ad9d245106828382490b8d2 Mon Sep 17 00:00:00 2001 From: Siarhei Yarkavy Date: Mon, 5 Oct 2026 17:11:39 +0300 Subject: [PATCH 5/8] fix: add gpt-6-astra to VS Code capabilities and OpenCode configs VS Code capability table: gpt-6-astra entry (922000/128000, Responses, GPT_6_EFFORTS) alongside luna/sol. Addresses review thread r4183491171 in #582. --- .../opencode/opencode-model-configs.ts | 29 +++++++++++++++++++ .../proxy/connectors/__tests__/vscode.test.ts | 2 ++ .../proxy/connectors/vscode-models.ts | 11 +++++++ 3 files changed, 42 insertions(+) diff --git a/src/agents/plugins/opencode/opencode-model-configs.ts b/src/agents/plugins/opencode/opencode-model-configs.ts index e87ac3418..85068943d 100644 --- a/src/agents/plugins/opencode/opencode-model-configs.ts +++ b/src/agents/plugins/opencode/opencode-model-configs.ts @@ -347,6 +347,35 @@ export const OPENCODE_MODEL_CONFIGS: Record = { output: 128000 } }, + 'gpt-6-astra': { + id: 'gpt-6-astra', + name: 'GPT-6 Astra', + displayName: 'GPT-6 Astra', + family: 'gpt-6', + tool_call: true, + reasoning: true, + attachment: true, + temperature: false, + structured_output: true, + use_responses_api: true, + modalities: { + input: ['text', 'image'], + output: ['text'] + }, + knowledge: '2026-05-18', + release_date: '2026-09-22', + last_updated: '2026-09-22', + open_weights: false, + cost: { + input: 10, + output: 50, + cache_read: 1 + }, + limit: { + context: 1050000, + output: 128000 + } + }, // ── Claude Models ────────────────────────────────────────────────── 'claude-4-5-sonnet': { diff --git a/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts b/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts index c7d5e94fb..ff551a3ef 100644 --- a/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts +++ b/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts @@ -27,6 +27,7 @@ const EXPECTED_MODEL_IDS = [ 'gpt-5.6-terra-2026-07-09', 'gpt-6-luna', 'gpt-6-sol', + 'gpt-6-astra', 'gemini-3-flash', 'gemini-3.1-pro', 'gemini-3.5-flash', @@ -184,6 +185,7 @@ describe('writeVsCodeLanguageModelsConfigAtPath', () => { ['gpt-5.6-terra-2026-07-09', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], ['gpt-6-luna', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], ['gpt-6-sol', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], + ['gpt-6-astra', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], ]); for (const [id, efforts] of expectedEfforts) { diff --git a/src/cli/commands/proxy/connectors/vscode-models.ts b/src/cli/commands/proxy/connectors/vscode-models.ts index 1348279f9..9ae86bf61 100644 --- a/src/cli/commands/proxy/connectors/vscode-models.ts +++ b/src/cli/commands/proxy/connectors/vscode-models.ts @@ -192,6 +192,17 @@ export const VS_CODE_CAPABILITY_TABLE: readonly VsCodeCapabilityEntry[] = [ maxInputTokens: 922000, maxOutputTokens: 128000, }, + { + family: 'gpt-6-astra', + apiType: 'responses', + vision: true, + thinking: true, + zeroDataRetentionEnabled: true, + supportsReasoningEffort: GPT_6_EFFORTS, + reasoningEffortFormat: 'responses', + maxInputTokens: 922000, + maxOutputTokens: 128000, + }, { family: 'gemini-3-flash', apiType: 'chat-completions', From c751f5c8a9a779bf66e490b26477fdbd0168d84a Mon Sep 17 00:00:00 2001 From: Siarhei Yarkavy Date: Mon, 5 Oct 2026 17:19:39 +0300 Subject: [PATCH 6/8] fix: add gpt-6.1-sol to VS Code capabilities and OpenCode configs VS Code capability table: gpt-6.1-sol entry (922000/128000, Responses, GPT_6_EFFORTS); dotted family key per gpt-5.6 precedent, resolves dashed, vendor-prefixed and dated ids. Static OpenCode config: 1050000/128000; cost and dates mirrored from gpt-6-sol pending vendor pricing verification (no pricing row for 6.1-sol yet; no 6.1-astra shipped). --- .../opencode/opencode-model-configs.ts | 29 +++++++++++++++++++ .../proxy/connectors/__tests__/vscode.test.ts | 2 ++ .../proxy/connectors/vscode-models.ts | 11 +++++++ 3 files changed, 42 insertions(+) diff --git a/src/agents/plugins/opencode/opencode-model-configs.ts b/src/agents/plugins/opencode/opencode-model-configs.ts index 85068943d..293fec55b 100644 --- a/src/agents/plugins/opencode/opencode-model-configs.ts +++ b/src/agents/plugins/opencode/opencode-model-configs.ts @@ -376,6 +376,35 @@ export const OPENCODE_MODEL_CONFIGS: Record = { output: 128000 } }, + 'gpt-6.1-sol': { + id: 'gpt-6.1-sol', + name: 'GPT-6.1 Sol', + displayName: 'GPT-6.1 Sol', + family: 'gpt-6', + tool_call: true, + reasoning: true, + attachment: true, + temperature: false, + structured_output: true, + use_responses_api: true, + modalities: { + input: ['text', 'image'], + output: ['text'] + }, + knowledge: '2026-05-18', + release_date: '2026-09-22', + last_updated: '2026-09-22', + open_weights: false, + cost: { + input: 2, + output: 10, + cache_read: 0.20 + }, + limit: { + context: 1050000, + output: 128000 + } + }, // ── Claude Models ────────────────────────────────────────────────── 'claude-4-5-sonnet': { diff --git a/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts b/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts index ff551a3ef..5e3256d80 100644 --- a/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts +++ b/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts @@ -28,6 +28,7 @@ const EXPECTED_MODEL_IDS = [ 'gpt-6-luna', 'gpt-6-sol', 'gpt-6-astra', + 'gpt-6.1-sol', 'gemini-3-flash', 'gemini-3.1-pro', 'gemini-3.5-flash', @@ -186,6 +187,7 @@ describe('writeVsCodeLanguageModelsConfigAtPath', () => { ['gpt-6-luna', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], ['gpt-6-sol', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], ['gpt-6-astra', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], + ['gpt-6.1-sol', ['none', 'low', 'medium', 'high', 'xhigh', 'max']], ]); for (const [id, efforts] of expectedEfforts) { diff --git a/src/cli/commands/proxy/connectors/vscode-models.ts b/src/cli/commands/proxy/connectors/vscode-models.ts index 9ae86bf61..a794e23b8 100644 --- a/src/cli/commands/proxy/connectors/vscode-models.ts +++ b/src/cli/commands/proxy/connectors/vscode-models.ts @@ -203,6 +203,17 @@ export const VS_CODE_CAPABILITY_TABLE: readonly VsCodeCapabilityEntry[] = [ maxInputTokens: 922000, maxOutputTokens: 128000, }, + { + family: 'gpt-6.1-sol', + apiType: 'responses', + vision: true, + thinking: true, + zeroDataRetentionEnabled: true, + supportsReasoningEffort: GPT_6_EFFORTS, + reasoningEffortFormat: 'responses', + maxInputTokens: 922000, + maxOutputTokens: 128000, + }, { family: 'gemini-3-flash', apiType: 'chat-completions', From b7892f93d81de457d7b083eb4e98fef04aedef2e Mon Sep 17 00:00:00 2001 From: Siarhei Yarkavy Date: Mon, 5 Oct 2026 17:29:28 +0300 Subject: [PATCH 7/8] fix: correct gpt-6.1-sol pricing from vendor Standard table Static OpenCode config cache_read 0.20 -> 0.1; add gpt-6-1-sol row to pricing.json (2/10/0.1/2.5, dashed key per file convention; dotted and dated ids resolve via normalizeModelName/snapshot). Source: https://developers.openai.com/api/docs/pricing?latest-pricing=standard (Standard pricing table, row 'gpt-6.1-sol'). Supersedes the mirrored-from-gpt-6-sol placeholder in c751f5c8. --- src/agents/plugins/opencode/opencode-model-configs.ts | 2 +- src/utils/pricing.json | 7 +++++++ 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/src/agents/plugins/opencode/opencode-model-configs.ts b/src/agents/plugins/opencode/opencode-model-configs.ts index 293fec55b..af5992bc6 100644 --- a/src/agents/plugins/opencode/opencode-model-configs.ts +++ b/src/agents/plugins/opencode/opencode-model-configs.ts @@ -398,7 +398,7 @@ export const OPENCODE_MODEL_CONFIGS: Record = { cost: { input: 2, output: 10, - cache_read: 0.20 + cache_read: 0.1 }, limit: { context: 1050000, diff --git a/src/utils/pricing.json b/src/utils/pricing.json index bac8eb008..c46f8283d 100644 --- a/src/utils/pricing.json +++ b/src/utils/pricing.json @@ -16,6 +16,7 @@ "keys": "All keys use dashes instead of dots (e.g. gpt-5-4 not gpt-5.4) so normalizeModelName dots→dashes works", "cacheRead": "cache hit price (Anthropic: 0.1x input, OpenAI: varies, Google: ~0.1x input)", "cacheWrite": "cache write price (Anthropic: 1.25x input, OpenAI: ~same as input)", + "gpt61Sol": "Row added 2026-10-05: gpt-6-1-sol short-context Standard $2.00/$0.10/$2.50/$10.00 verified against https://developers.openai.com/api/docs/pricing?latest-pricing=standard (Standard pricing table, row 'gpt-6.1-sol'). Dashed key per this file's convention; dotted ids normalize via normalizeModelName.", "claude5": "Sonnet 5 and Opus 5 standard API-equivalent token rates verified 2026-09-15; Fable 5/5.1, Mythos 5/5.1 and Opus 4.7/4.8 verified 2026-09-27 at https://platform.claude.com/docs/en/about-claude/pricing; excludes additional tool charges. Sonnet 5's planned September 1 price increase was cancelled.", "t10CatalogReconciliation": "Rows added 2026-09-26 to price catalog ids that resolvePrice() found unpriced (unify-analytics-cost-command T10), each verified against one official vendor page/API on that date. claude-opus-5-5: $4/$20 base, $5/$8 5m/1h cache write, $0.20 cache read (0.05x multiplier), bedrockRegionalMultiplier 1.1 (Opus 4.5+ scope) — https://platform.claude.com/docs/en/about-claude/pricing (Model pricing table row 'Claude Opus 5.5'). gpt-5-5: short-context Standard row '$5.00 | $0.50 | - | $30.00' — https://developers.openai.com/api/docs/pricing ('Standard Pricing Tables', row 'gpt-5.5 (<272K context length)'); cache-write column is '-' (not stated) so cacheWrite mirrors input, matching this table's existing gpt-5-4 handling. gpt-5-6-luna/-sol/-terra and gpt-6-luna/-sol: short-context Standard rows from the same OpenAI page/table, e.g. 'gpt-6-sol | $2.00 | $0.20 | $2.50 | $10.00'; these rows do state a real cache-write price (~1.25x input) so it is recorded as-is rather than mirrored. deepseek-v4-pro: peak (weekday 01:00-04:00/06:00-10:00 UTC) cache-miss input $1.32, cache-hit input $0.044, output $3.96 — https://api-docs.deepseek.com/quick_start/pricing/; off-peak (all other hours, including weekends) is exactly half: $0.66/$0.022/$1.98; peak chosen as the flat row like every other entry in this table, no time-of-day tiering supported by the schema; cacheWrite mirrors input per this table's deepseek-chat/deepseek-reasoner precedent. gemini-3-6-flash/gemini-3-8-flash: paid-tier Standard pricing 'through Dec 31, 2026' — $0.75 input/$3.75 output/$0.075 context-caching — https://ai.google.dev/gemini-api/docs/pricing; scheduled to double to $1.50/$7.50/$0.15 on 2027-01-01, not yet in effect as of verification; cacheWrite mirrors input per this table's gemini-3-5-flash/gemini-3-7-flash precedent. gemini-3-1-flash-image: Standard pricing 'Input: $0.50 (text/image); Output: $3 (text/thinking), $60.00 (images)' — https://ai.google.dev/gemini-api/docs/pricing; output field records the $60 image-generation rate (not the $3 text/thinking rate), matching this table's existing gemini-3-1-flash-image-preview/gemini-2-5-flash-image rows, which use the same convention (schema has no separate image-output field); cacheRead/cacheWrite 0 (no cache pricing published for this model). grok-4-6: base tier (<200k prompt tokens) row 'grok-4.6 (< 200k prompt tokens) | 500k | $2.00 | $0.50 | $6.00' — https://docs.x.ai/developers/models; xAI doubles every rate for prompts >=200k tokens, not represented here (schema is flat), so this is the base/short-context tier, consistent with this table's other short-context-only entries; cacheWrite mirrors input per this table's grok-4/grok-4-1-fast precedent. qwen3-coder-30b-a3b-v1 and qwen3-coder-480b-a35b-v1: AWS Bedrock on-demand Standard tier, US East (N. Virginia) — official AWS Price List API https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/20260926004940/us-east-1/index.json (also rendered at https://aws.amazon.com/bedrock/pricing/), SKU usagetype 'qwen.qwen3-coder-30b-a3b-instruct-mantle-{input,output}-tokens-standard' = $0.00015/$0.0006 per 1K tokens ($0.15/$0.60 per 1M) and 'qwen.qwen3-coder-480b-a35b-instruct-mantle-{input,output}-tokens-standard' = $0.00045/$0.0018 per 1K tokens ($0.45/$1.80 per 1M); no cache pricing published for either. Checked but NOT added (no official page states a price for the exact catalog id): claude-4-5-sonnet / claude-4-5-sonnet-vertex now resolve via the T1 reorder fallback onto the existing claude-sonnet-4-5 row, so no new row was needed." }, @@ -267,6 +268,12 @@ "cacheRead": 0.2, "cacheWrite": 2.5 }, + "gpt-6-1-sol": { + "input": 2, + "output": 10, + "cacheRead": 0.1, + "cacheWrite": 2.5 + }, "gpt-5-2": { "input": 1.75, "output": 14, From d08fb4eed1ee3f04f87ff1f43d6206c3406791c9 Mon Sep 17 00:00:00 2001 From: Siarhei Yarkavy Date: Mon, 5 Oct 2026 17:31:24 +0300 Subject: [PATCH 8/8] fix: correct gpt-6-astra cacheWrite from vendor Standard table cacheWrite 10 -> 12.5 per https://developers.openai.com/api/docs/pricing?latest-pricing=standard (Standard pricing table, row 'gpt-6-astra': short-context 0.00/.00/2.50/0.00). 10 was a mirrored-input error carried over from #583; real price is ~1.25x input like the other GPT-6 rows. --- src/utils/pricing.json | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/src/utils/pricing.json b/src/utils/pricing.json index c46f8283d..30501037f 100644 --- a/src/utils/pricing.json +++ b/src/utils/pricing.json @@ -17,6 +17,7 @@ "cacheRead": "cache hit price (Anthropic: 0.1x input, OpenAI: varies, Google: ~0.1x input)", "cacheWrite": "cache write price (Anthropic: 1.25x input, OpenAI: ~same as input)", "gpt61Sol": "Row added 2026-10-05: gpt-6-1-sol short-context Standard $2.00/$0.10/$2.50/$10.00 verified against https://developers.openai.com/api/docs/pricing?latest-pricing=standard (Standard pricing table, row 'gpt-6.1-sol'). Dashed key per this file's convention; dotted ids normalize via normalizeModelName.", + "astraCacheWriteFix": "Corrected 2026-10-05: gpt-6-astra cacheWrite 10 -> 12.5 per the same vendor Standard table (row 'gpt-6-astra': $10.00/$1.00/$12.50/$50.00); 10 was a mirrored-input error, real price is ~1.25x input.", "claude5": "Sonnet 5 and Opus 5 standard API-equivalent token rates verified 2026-09-15; Fable 5/5.1, Mythos 5/5.1 and Opus 4.7/4.8 verified 2026-09-27 at https://platform.claude.com/docs/en/about-claude/pricing; excludes additional tool charges. Sonnet 5's planned September 1 price increase was cancelled.", "t10CatalogReconciliation": "Rows added 2026-09-26 to price catalog ids that resolvePrice() found unpriced (unify-analytics-cost-command T10), each verified against one official vendor page/API on that date. claude-opus-5-5: $4/$20 base, $5/$8 5m/1h cache write, $0.20 cache read (0.05x multiplier), bedrockRegionalMultiplier 1.1 (Opus 4.5+ scope) — https://platform.claude.com/docs/en/about-claude/pricing (Model pricing table row 'Claude Opus 5.5'). gpt-5-5: short-context Standard row '$5.00 | $0.50 | - | $30.00' — https://developers.openai.com/api/docs/pricing ('Standard Pricing Tables', row 'gpt-5.5 (<272K context length)'); cache-write column is '-' (not stated) so cacheWrite mirrors input, matching this table's existing gpt-5-4 handling. gpt-5-6-luna/-sol/-terra and gpt-6-luna/-sol: short-context Standard rows from the same OpenAI page/table, e.g. 'gpt-6-sol | $2.00 | $0.20 | $2.50 | $10.00'; these rows do state a real cache-write price (~1.25x input) so it is recorded as-is rather than mirrored. deepseek-v4-pro: peak (weekday 01:00-04:00/06:00-10:00 UTC) cache-miss input $1.32, cache-hit input $0.044, output $3.96 — https://api-docs.deepseek.com/quick_start/pricing/; off-peak (all other hours, including weekends) is exactly half: $0.66/$0.022/$1.98; peak chosen as the flat row like every other entry in this table, no time-of-day tiering supported by the schema; cacheWrite mirrors input per this table's deepseek-chat/deepseek-reasoner precedent. gemini-3-6-flash/gemini-3-8-flash: paid-tier Standard pricing 'through Dec 31, 2026' — $0.75 input/$3.75 output/$0.075 context-caching — https://ai.google.dev/gemini-api/docs/pricing; scheduled to double to $1.50/$7.50/$0.15 on 2027-01-01, not yet in effect as of verification; cacheWrite mirrors input per this table's gemini-3-5-flash/gemini-3-7-flash precedent. gemini-3-1-flash-image: Standard pricing 'Input: $0.50 (text/image); Output: $3 (text/thinking), $60.00 (images)' — https://ai.google.dev/gemini-api/docs/pricing; output field records the $60 image-generation rate (not the $3 text/thinking rate), matching this table's existing gemini-3-1-flash-image-preview/gemini-2-5-flash-image rows, which use the same convention (schema has no separate image-output field); cacheRead/cacheWrite 0 (no cache pricing published for this model). grok-4-6: base tier (<200k prompt tokens) row 'grok-4.6 (< 200k prompt tokens) | 500k | $2.00 | $0.50 | $6.00' — https://docs.x.ai/developers/models; xAI doubles every rate for prompts >=200k tokens, not represented here (schema is flat), so this is the base/short-context tier, consistent with this table's other short-context-only entries; cacheWrite mirrors input per this table's grok-4/grok-4-1-fast precedent. qwen3-coder-30b-a3b-v1 and qwen3-coder-480b-a35b-v1: AWS Bedrock on-demand Standard tier, US East (N. Virginia) — official AWS Price List API https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/20260926004940/us-east-1/index.json (also rendered at https://aws.amazon.com/bedrock/pricing/), SKU usagetype 'qwen.qwen3-coder-30b-a3b-instruct-mantle-{input,output}-tokens-standard' = $0.00015/$0.0006 per 1K tokens ($0.15/$0.60 per 1M) and 'qwen.qwen3-coder-480b-a35b-instruct-mantle-{input,output}-tokens-standard' = $0.00045/$0.0018 per 1K tokens ($0.45/$1.80 per 1M); no cache pricing published for either. Checked but NOT added (no official page states a price for the exact catalog id): claude-4-5-sonnet / claude-4-5-sonnet-vertex now resolve via the T1 reorder fallback onto the existing claude-sonnet-4-5 row, so no new row was needed." }, @@ -242,7 +243,7 @@ "input": 10, "output": 50, "cacheRead": 1, - "cacheWrite": 10 + "cacheWrite": 12.5 }, "gpt-5-5-pro": { "input": 30,