diff --git a/docs/COMMANDS.md b/docs/COMMANDS.md index 1f92679b0..b1e625717 100644 --- a/docs/COMMANDS.md +++ b/docs/COMMANDS.md @@ -105,7 +105,7 @@ codemie proxy connect vscode --profile work codemie proxy connect vscode --insiders ``` -The connector resolves the selected profile once, synchronizes skills, and writes every enabled tenant model from the live catalog into VS Code's `User/chatLanguageModels.json`, in catalog order. Known families are enriched from the capability table; unknown models get conservative defaults rather than being dropped. The profile's default model does not affect which models are written. VS Code sends the configured model ID directly; the proxy authenticates the request, adds CodeMie context headers, and applies only the documented compatibility normalization before forwarding. +The connector resolves the selected profile once, synchronizes skills, and writes every enabled tenant model from the live catalog into VS Code's `User/chatLanguageModels.json`, in catalog order. Known families are enriched from the capability table; unknown models get conservative defaults rather than being dropped. The output limit (`maxOutputTokens`) resolves in the order tenant catalog, built-in capability table, default (8192). The input limit (`maxInputTokens`) is the catalog `max_input_tokens` minus the resolved output limit, so input plus output fits the context window; when the catalog has no usable input value, or the subtraction is not positive, it is the table value or 128000. The profile's default model does not affect which models are written. VS Code sends the configured model ID directly; the proxy authenticates the request, adds CodeMie context headers, and applies only the documented compatibility normalization before forwarding. `--profile ` is a one-command override and does not change the active CodeMie profile. Model and project remain independent: the model is written into VS Code configuration, while `codeMieProject` is passed to the daemon and emitted as `X-CodeMie-Project`. When a selected profile has no project of its own, compatible repository-local project context continues to apply through the standard profile merge rules. diff --git a/docs/stories/2026-10-02-vscode-model-token-limits/complexity-assessment.json b/docs/stories/2026-10-02-vscode-model-token-limits/complexity-assessment.json new file mode 100644 index 000000000..5a69d4f10 --- /dev/null +++ b/docs/stories/2026-10-02-vscode-model-token-limits/complexity-assessment.json @@ -0,0 +1,30 @@ +{ + "schema": 1, + "task": "Make codemie proxy connect --vscode resolve model token limits from the tenant catalog API first, then the built-in capability table, then defaults.", + "generated": "2026-10-02T00:00:00Z", + "dimensions": { + "component_scope": { "score": 3, "label": "M" }, + "requirements_clarity": { "score": 2, "label": "S" }, + "technical_risk": { "score": 2, "label": "S" }, + "file_change_estimate": { "score": 3, "label": "M" }, + "dependencies": { "score": 1, "label": "XS" }, + "affected_layers": { "score": 2, "label": "S" } + }, + "total": 13, + "size": "S", + "size_legend": { + "XS": "6-9 — < half day — plan directly", + "S": "10-14 — 1 day — plan directly", + "M": "15-20 — 2-3 days — brainstorm first", + "L": "21-26 — 4-5 days — brainstorm first", + "XL": "27-31 — > 1 sprint — recommend splitting", + "XXL": "32-36 — > 1 sprint — must split" + }, + "routing": "writing-plans", + "key_reasoning": [ + { "dimension": "component_scope", "reason": "Three to four components (tenant-catalog parsing, vscode-models table, vscode connector assembly, SSO http-client type) in one repo, with a clear per-field fallback pattern. Scored M, below L. No project calibration example is near this band; the only one is an L-band example." }, + { "dimension": "file_change_estimate", "reason": "About 4 source files plus 1 doc modified, no new files, so M." } + ], + "red_flags_applied": [], + "split_recommendation": null +} diff --git a/docs/stories/2026-10-02-vscode-model-token-limits/story.md b/docs/stories/2026-10-02-vscode-model-token-limits/story.md new file mode 100644 index 000000000..14339380b --- /dev/null +++ b/docs/stories/2026-10-02-vscode-model-token-limits/story.md @@ -0,0 +1,64 @@ +# VS Code models get too-low token limits — Story + +**Date**: 2026-10-02 +**Status**: Approved +**Type**: Bug +**Ticket**: [EPMCDME-15572](https://jiraeu.epam.com/browse/EPMCDME-15572) + +--- + +## Context + +- `codemie proxy connect --vscode` creates one VS Code model entry per enabled tenant model, each with an input and an output token limit. +- Today those limits come only from a built-in table of known models, with a generic default for everything else. +- The tenant model catalog the command already reads reports the input limit for most models, but it is ignored. +- Models missing from the built-in table fall back to 128000 even when the tenant reports up to 1M. +- The output limit is not returned by the catalog yet, so it keeps coming from the table or default for now. + +--- + +## Complexity + +**Size**: S (13/36) · **Recommended flow**: sdlc-light + +Small, contained change in a single repository; plan directly. + +--- + +## Story + +**As a** developer using CodeMie models in VS Code, **I want** each model's token limits to reflect what the tenant actually reports **so that** long conversations aren't summarized or truncated earlier than necessary. + +--- + +## Background + +VS Code uses the configured input limit as its prompt budget. Newer models (large-context GPT, Gemini, Claude, DeepSeek, Grok) are configured with 128000 although the tenant reports 500k–1M, so context is cut far earlier than needed. Limits should come from the tenant first, then the built-in table, then the generic defaults. The catalog input value may be the whole context window, so the resolved output limit is subtracted from it. + +--- + +## Acceptance Criteria + +- [ ] Given the tenant catalog reports an input limit for a model, when VS Code models are generated, then the input limit is that value minus the resolved output limit, even if the built-in table has a different one. +- [ ] Given the tenant catalog reports an output limit for a model, when VS Code models are generated, then that value is used as the model's output limit, and the input limit subtracts that value. +- [ ] Given the catalog input limit minus the resolved output limit is zero or negative, when VS Code models are generated, then the built-in table value (or the default) is used for the input limit. +- [ ] Given the catalog has no usable input limit for a model that exists in the built-in table, when VS Code models are generated, then the table value is used. +- [ ] Given the catalog has no usable input limit for a model that is not in the table, when VS Code models are generated, then the default value is used. +- [ ] Given the catalog value for a limit is missing, zero, negative or not a number, when VS Code models are generated, then it is ignored and the next source is used. +- [ ] Given the input and output limits come from different sources, when a model is generated, then the output limit is resolved from catalog, table, then default, and the input limit depends on the resolved output limit but falls back to the table or default independently when the catalog gives no usable input value. +- [ ] Given a tenant that reports no token limits at all, when VS Code models are generated, then results are identical to today's. +- [ ] Given the change is released, when a user reads the VS Code section of the command documentation, then it states the order (tenant catalog, built-in table, defaults) and that the output limit is subtracted from the catalog input limit. + +--- + +## Out of Scope + +- Changing the backend to return an output limit. +- Moving reasoning effort, API type or header settings out of the built-in table. +- Removing existing built-in table values. + +--- + +## Open Questions + +- None. diff --git a/docs/stories/2026-10-02-vscode-model-token-limits/technical-analysis.md b/docs/stories/2026-10-02-vscode-model-token-limits/technical-analysis.md new file mode 100644 index 000000000..ac848abeb --- /dev/null +++ b/docs/stories/2026-10-02-vscode-model-token-limits/technical-analysis.md @@ -0,0 +1,39 @@ +# Technical Analysis — VS Code model token limits + +**Date**: 2026-10-02 +**Source**: `VSCODE_MODEL_TOKEN_LIMITS.md` (reviewed against the proposal; no separate codebase sweep) + +## Feature area + +`codemie proxy connect --vscode` — generation of VS Code BYOK model entries (`chatLanguageModels.json`). + +## Current behaviour + +- Each enabled tenant model gets an entry with `maxInputTokens` and `maxOutputTokens`. +- Both values come from the repository: a hardcoded capability table for known model families, and fixed defaults (128000 input / 8192 output) for everything else. +- The connector already fetches the tenant model catalog (`GET /v1/llm_models?include_all=true`), which reports `max_input_tokens` for most models, but the value is ignored when building entries. +- Effect: models absent from the table get 128000 even when the tenant reports up to ~1M; VS Code uses the value as its prompt budget and summarizes context too early. + +## Affected components + +- Tenant catalog parsing (`src/cli/commands/proxy/connectors/tenant-catalog.ts`) — model descriptor does not carry token limits. +- VS Code model capability table and defaults (`src/cli/commands/proxy/connectors/vscode-models.ts`). +- VS Code connector model assembly (`src/cli/commands/proxy/connectors/vscode.ts`). +- SSO HTTP client model type (`src/providers/plugins/sso/sso.http-client.ts`) — no `max_output_tokens` field yet. +- User docs (`docs/COMMANDS.md`, VS Code BYOK section). + +## Proposed resolution order (per field, independently) + +| Field | 1st | 2nd | 3rd | +|---|---|---|---| +| Input limit | tenant catalog value | capability table | default 128000 | +| Output limit | tenant catalog value (not returned by API today) | capability table | default 8192 | + +A value is usable only if it is a finite number > 0; otherwise fall through. + +## Risks / notes + +- Output limits remain table/default-driven until the backend returns `max_output_tokens`. +- Models with no catalog value (routers, static-config catalogs) keep today's behaviour. +- Reasoning efforts, API type and headers stay in the table (no API equivalent). +- Small, contained change: ~4 source files + 1 doc, single repository, no new dependencies. diff --git a/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/actual-complexity.json b/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/actual-complexity.json new file mode 100644 index 000000000..4530814e6 --- /dev/null +++ b/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/actual-complexity.json @@ -0,0 +1,23 @@ +{ + "schema": 1, + "generated": "2026-10-02T00:00:00Z", + "dimensions": { + "component_scope": { "score": 3, "label": "M" }, + "requirements_clarity": { "score": 2, "label": "S" }, + "technical_risk": { "score": 2, "label": "S" }, + "file_change_estimate": { "score": 3, "label": "M" }, + "dependencies": { "score": 1, "label": "XS" }, + "affected_layers": { "score": 2, "label": "S" } + }, + "total": 13, + "size": "S", + "band_range": "10-14", + "files_changed": 5, + "routing": "writing-plans", + "key_reasoning": [ + { "dimension": "component_scope", "reason": "Touches tenant-catalog descriptor parsing, the vscode-models capability table/resolver, the vscode connector, and the LlmModel type in sso.http-client; pattern is clear and mirrors existing descriptor fields." }, + { "dimension": "file_change_estimate", "reason": "5 files changed (4 source, 1 doc), 50 insertions and 7 deletions, no new files." } + ], + "red_flags_applied": [], + "split_recommendation": null +} diff --git a/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/code-review-brief.md b/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/code-review-brief.md new file mode 100644 index 000000000..d4e4ddee4 --- /dev/null +++ b/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/code-review-brief.md @@ -0,0 +1,12 @@ +# Code review — 2026-10-02-vscode-model-token-limits (2026-10-05) + +**approve** · confidence: low · 0 blocking · 1 deferred · 2 filtered as noise +Coverage: blind — n/a (compact profile) · edge-case ✓ · verification-gap — n/a (compact profile) · acceptance — n/a (no spec) (1/1 applicable lenses ran) + +## Look here first + +No blocking findings — the diff speaks for itself. + +## Checked and clean + +commit-format — n/a · code-quality — n/a · security — n/a (standards not run for this profile) · 1 deferred → code-review-deferred.md diff --git a/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/code-review-final.json b/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/code-review-final.json new file mode 100644 index 000000000..949037b44 --- /dev/null +++ b/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/code-review-final.json @@ -0,0 +1 @@ +{"decision":"approve","rationale":"Only the edge-case lens ran (compact profile, no spec so acceptance skipped), hence low confidence. Two edge findings dismissed as spec-prescribed behavior (fallback to entry value when subtraction is not positive; catalog input treated as full context window); 1 deferred (fractional token limits accepted by pre-existing validator) listed in code-review-deferred.md.","confidence":"low","risk_flags":[],"business_review":[],"standards_review":[{"kind":"commit-format","status":"na","notes":"Standards audit not expected for this profile."},{"kind":"code-quality","status":"na","notes":"Standards audit not expected for this profile."},{"kind":"security","status":"na","notes":"Standards audit not expected for this profile."}],"findings":[]} diff --git a/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/plan.md b/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/plan.md new file mode 100644 index 000000000..0ffbf9388 --- /dev/null +++ b/docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/plan.md @@ -0,0 +1,63 @@ +# VS Code Model Token Limits: Subtract Output From Catalog Input + +> **For agentic workers:** Use superpowers:subagent-driven-development or superpowers:executing-plans. + +**Goal:** In `codemie proxy connect --vscode`, `maxInputTokens` = catalog `max_input_tokens` minus the resolved `maxOutputTokens`, so input plus output fits the context window. + +**Requirements:** `docs/VSCODE_TOKEN_LIMITS_ADJUSTMENT.md` (authoritative). Research: `technical-analysis.md` in this dir. + +**Already on the branch (do not redo):** catalog parsing of `max_input_tokens`/`max_output_tokens` in `tenant-catalog.ts`; `LlmModel.max_output_tokens`; `resolveVsCodeTokenLimits` + `pickTokenLimit` in `vscode-models.ts:385-403` (currently returns the catalog input unchanged); `vscode.ts:119` already uses the resolver for table and default models. Only the input rule, comments, docs, task records and tests change. + +**Commits:** Commit per task using the repository's existing convention (Conventional Commits). Do not commit `.codemie/codemie-cli.config.json`; leave other dirty files alone. + +## Acceptance criteria + +- Catalog `max_input_tokens` present: written `maxInputTokens` = that value minus the resolved `maxOutputTokens` (catalog, then table, then 8192). +- Subtraction result <= 0, or no usable catalog input: entry value (table value or 128000). +- Output limit order unchanged: catalog, table, 8192; input and output resolve independently. +- Tenant reporting no limits yields today's output. +- `docs/COMMANDS.md` describes the rule; example `gpt-5.6-sol` shows 922000. +- Story and task-dir records match the shipped rule. +- New tests in the three connector test files pass; lint and typecheck pass. + +Negative-constraints pass: do not leave catalog input unchanged (Task 1); do not remove table values or change `tenant-catalog.ts`/`vscode.ts` (not touched); do not test every table family or repeat the fallback order across files (Task 3 scope); the doc's "Don't" items honored; ignore `.codemie/codemie-cli.config.json` (header). + +--- + +### Task 1: Subtract output from catalog input, update comments + +**Files:** +- Modify: `src/cli/commands/proxy/connectors/vscode-models.ts` (`resolveVsCodeTokenLimits` ~395-403 and its doc comment; comment above `VS_CODE_CAPABILITY_TABLE` ~48-51) +- Modify: `src/providers/plugins/sso/sso.http-client.ts:196-200` (doc comment on `max_input_tokens`) + +Test-first: yes — `resolveVsCodeTokenLimits` with descriptor `maxInputTokens: 200000` and Claude 4.5 table entry (output 64000) expects `maxInputTokens` 136000 (fails today: returns 200000). + +- [ ] Resolve `maxOutputTokens` first via `pickTokenLimit`. Then if the descriptor's input passes the same positive-finite check, use `input - maxOutputTokens` when that is > 0, else `entry.maxInputTokens`. Rewrite the function doc: API input is treated as the whole context window, output is subtracted so the pair fits. +- [ ] Table comment: keep the "fallbacks" note; add that the table's `maxInputTokens` is a prompt budget (window minus output) while the catalog value is not, hence the subtraction. In `sso.http-client.ts`, say the value is the whole window for some models and only the prompt budget for others, so callers must not assume it fits alongside the output limit. + +### Task 2: Docs and task records + +**Files:** +- Modify: `docs/COMMANDS.md` (paragraph ~108; JSON example ~155) +- Modify: `docs/stories/2026-10-02-vscode-model-token-limits/story.md` (line 36 background; line 42 first criterion) +- Modify: `docs/superpowers/tasks/2026-10-02-vscode-model-token-limits/technical-analysis.md` (minimal edits only where it states the old rule: Section 1 "used as-is"/ordering text, Section 6 risk bullet about larger `maxInputTokens`, Section 7 summary, Section 8 key facts "no arithmetic") + +Test-first: no — documentation only + +- [ ] `COMMANDS.md`: input limit is the catalog value minus the resolved output limit, else table value or 128000; output is catalog, table, 8192; keep the note that the catalog does not return an output limit yet. Change the `gpt-5.6-sol` example `maxInputTokens` from 1050000 to 922000. +- [ ] `story.md`: line 36 explains the catalog input may be the whole window so the output limit is subtracted; line 42 says input = catalog value minus resolved output limit and still wins over the table. Leave line 48 as is. +- [ ] `technical-analysis.md`: change the old "unchanged/as-is, no arithmetic" statements and the 922000 -> 1050000 example to the subtract rule, and reword the risk bullet as mitigated by subtraction. Keep other content untouched; this plan already replaces the old one. + +### Task 3: Tests + +**Files:** +- Modify: `src/cli/commands/proxy/connectors/__tests__/vscode-models.test.ts` +- Modify: `src/cli/commands/proxy/connectors/__tests__/tenant-catalog.test.ts` (next to the "maps label, provider, multimodal and features.tools" test, ~116-176) +- Modify: `src/cli/commands/proxy/connectors/__tests__/vscode.test.ts` (using `writeVsCodeLanguageModelsConfigAtPath`; existing fixtures stay unchanged) + +Test-first: yes — each new case asserts the new rule or parsing (e.g. untabled model with catalog input 922000 expects written `maxInputTokens` 913808, `maxOutputTokens` 8192; fails before Task 1). + +- [ ] `vscode-models.test.ts`, `describe('resolveVsCodeTokenLimits')` with real table numbers: catalog input + table output (Claude 4.5, 200000 -> 136000); catalog input + default output (untabled, API value - 8192); catalog input + catalog output (subtract catalog output); catalog output overrides table and default; subtraction <= 0 falls back to entry `maxInputTokens`; no catalog values returns entry unchanged; catalog output only (no input) keeps entry input. +- [ ] `tenant-catalog.test.ts`: valid `max_input_tokens`/`max_output_tokens` become `maxInputTokens`/`maxOutputTokens`; `0`, negative, string and `null` are omitted; missing fields omitted. +- [ ] `vscode.test.ts`: one fixture with `max_input_tokens` on one table family and one unknown model; assert both written `maxInputTokens` and `maxOutputTokens` in `chatLanguageModels.json`, proving the descriptor reaches table and default entries. +- [ ] Run `npx vitest run src/cli/commands/proxy/connectors/__tests__/`, `npm run lint`, `npm run typecheck` for the touched files' sake (these are task-level checks; the flow runs the full gates). diff --git a/src/cli/commands/proxy/connectors/__tests__/tenant-catalog.test.ts b/src/cli/commands/proxy/connectors/__tests__/tenant-catalog.test.ts index 7f0e8575a..fc3899213 100644 --- a/src/cli/commands/proxy/connectors/__tests__/tenant-catalog.test.ts +++ b/src/cli/commands/proxy/connectors/__tests__/tenant-catalog.test.ts @@ -142,6 +142,28 @@ describe('fetchTenantModelDescriptors', () => { ]); }); + it('maps valid max_input_tokens and max_output_tokens to token limits', async () => { + mockJson([{ base_name: 'm', max_input_tokens: 200000, max_output_tokens: 16000 }]); + + const descriptors = await fetchTenantModelDescriptors('http://127.0.0.1:4001', 'gw-key'); + expect(descriptors).toEqual([{ id: 'm', maxInputTokens: 200000, maxOutputTokens: 16000 }]); + }); + + it('omits zero, negative, string, null and missing token limits', async () => { + mockJson([ + { base_name: 'zero', max_input_tokens: 0, max_output_tokens: 0 }, + { base_name: 'negative', max_input_tokens: -5, max_output_tokens: -1 }, + { base_name: 'string', max_input_tokens: '200000', max_output_tokens: '16000' }, + { base_name: 'null', max_input_tokens: null, max_output_tokens: null }, + { base_name: 'missing' }, + ]); + + const descriptors = await fetchTenantModelDescriptors('http://127.0.0.1:4001', 'gw-key'); + expect(descriptors).toEqual([ + { id: 'zero' }, { id: 'negative' }, { id: 'string' }, { id: 'null' }, { id: 'missing' }, + ]); + }); + it('drops enabled: false entries and treats a missing enabled as enabled', async () => { mockJson({ data: [ { base_name: 'on', enabled: true }, diff --git a/src/cli/commands/proxy/connectors/__tests__/vscode-models.test.ts b/src/cli/commands/proxy/connectors/__tests__/vscode-models.test.ts index 00a4dc7ca..488f6b8bf 100644 --- a/src/cli/commands/proxy/connectors/__tests__/vscode-models.test.ts +++ b/src/cli/commands/proxy/connectors/__tests__/vscode-models.test.ts @@ -3,6 +3,7 @@ import { VS_CODE_CAPABILITY_TABLE, buildDefaultVsCodeCapability, findVsCodeCapabilityEntry, + resolveVsCodeTokenLimits, } from '../vscode-models.js'; describe('VS_CODE_CAPABILITY_TABLE', () => { @@ -89,3 +90,43 @@ describe('buildDefaultVsCodeCapability', () => { expect(entry.zeroDataRetentionEnabled).toBeUndefined(); }); }); + +describe('resolveVsCodeTokenLimits', () => { + const claude45 = findVsCodeCapabilityEntry('claude-4-5-sonnet')!; + const untabled = buildDefaultVsCodeCapability({ id: 'gpt-6-sol' }); + + it('subtracts the table output limit from catalog input', () => { + expect(resolveVsCodeTokenLimits(claude45, { id: 'claude-4-5-sonnet', maxInputTokens: 200000 })) + .toEqual({ maxInputTokens: 136000, maxOutputTokens: 64000 }); + }); + + it('subtracts the default output limit when the model is untabled', () => { + expect(resolveVsCodeTokenLimits(untabled, { id: 'gpt-6-sol', maxInputTokens: 922000 })) + .toEqual({ maxInputTokens: 913808, maxOutputTokens: 8192 }); + }); + + it('subtracts the catalog output limit when both are reported', () => { + expect(resolveVsCodeTokenLimits(untabled, { id: 'gpt-6-sol', maxInputTokens: 200000, maxOutputTokens: 16000 })) + .toEqual({ maxInputTokens: 184000, maxOutputTokens: 16000 }); + }); + + it('lets a catalog output limit override the table value', () => { + expect(resolveVsCodeTokenLimits(claude45, { id: 'claude-4-5-sonnet', maxInputTokens: 200000, maxOutputTokens: 32000 })) + .toEqual({ maxInputTokens: 168000, maxOutputTokens: 32000 }); + }); + + it('falls back to the entry input when the subtraction is not positive', () => { + expect(resolveVsCodeTokenLimits(claude45, { id: 'claude-4-5-sonnet', maxInputTokens: 64000 })) + .toEqual({ maxInputTokens: claude45.maxInputTokens, maxOutputTokens: 64000 }); + }); + + it('returns the entry unchanged when the catalog reports no limits', () => { + expect(resolveVsCodeTokenLimits(claude45, { id: 'claude-4-5-sonnet' })) + .toEqual({ maxInputTokens: claude45.maxInputTokens, maxOutputTokens: claude45.maxOutputTokens }); + }); + + it('keeps the entry input when only the catalog output is reported', () => { + expect(resolveVsCodeTokenLimits(claude45, { id: 'claude-4-5-sonnet', maxOutputTokens: 32000 })) + .toEqual({ maxInputTokens: claude45.maxInputTokens, maxOutputTokens: 32000 }); + }); +}); diff --git a/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts b/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts index 85d537b09..97074f75f 100644 --- a/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts +++ b/src/cli/commands/proxy/connectors/__tests__/vscode.test.ts @@ -326,6 +326,22 @@ describe('writeVsCodeLanguageModelsConfigAtPath', () => { expect(await readFile(configPath, 'utf-8')).toBe(original); }); + it('applies catalog token limits to both table and default entries', async () => { + mockCatalog([ + { base_name: 'claude-4-5-sonnet', max_input_tokens: 200000 }, + { base_name: 'gpt-6-sol', max_input_tokens: 922000 }, + ]); + + await writeVsCodeLanguageModelsConfigAtPath(configPath, 'http://127.0.0.1:4001', 'gw-key'); + + const providers = await readProviders(); + const models = providers[0].models as Array>; + expect(models.find(m => m.id === 'claude-4-5-sonnet')) + .toMatchObject({ maxInputTokens: 136000, maxOutputTokens: 64000 }); + expect(models.find(m => m.id === 'gpt-6-sol')) + .toMatchObject({ maxInputTokens: 913808, maxOutputTokens: 8192 }); + }); + describe('tenant-aware resolution against a non-EPAM-shaped catalog', () => { it('AC1: omits a capability-table family with no match in a sparse tenant catalog', async () => { mockCatalog(NON_EPAM_TENANT_FIXTURE); diff --git a/src/cli/commands/proxy/connectors/tenant-catalog.ts b/src/cli/commands/proxy/connectors/tenant-catalog.ts index e6b25ffa1..022e270dd 100644 --- a/src/cli/commands/proxy/connectors/tenant-catalog.ts +++ b/src/cli/commands/proxy/connectors/tenant-catalog.ts @@ -1,8 +1,9 @@ import { ConfigurationError } from '@/utils/errors.js'; import { logger } from '@/utils/logger.js'; import { sanitizeLogArgs } from '@/utils/security.js'; +import type { LlmModel } from '@/providers/plugins/sso/sso.http-client.js'; -interface CodeMieLlmModel { +interface CodeMieLlmModel extends Partial> { id?: string; base_name?: string; deployment_name?: string; @@ -22,6 +23,10 @@ export interface TenantModelDescriptor { multimodal?: boolean; /** From `features.tools`. */ toolCalling?: boolean; + /** From `max_input_tokens`; set only when a finite number > 0. */ + maxInputTokens?: number; + /** From `max_output_tokens`; set only when a finite number > 0. */ + maxOutputTokens?: number; } interface ModelsListResponse { @@ -50,12 +55,18 @@ function extractModelId(model: CodeMieLlmModel): string | undefined { return model.id || model.base_name || model.deployment_name; } +function isPositiveFiniteNumber(value: unknown): value is number { + return typeof value === 'number' && Number.isFinite(value) && value > 0; +} + function toDescriptor(model: CodeMieLlmModel, id: string): TenantModelDescriptor { const descriptor: TenantModelDescriptor = { id }; if (typeof model.label === 'string') descriptor.label = model.label; if (typeof model.provider === 'string') descriptor.provider = model.provider; if (typeof model.multimodal === 'boolean') descriptor.multimodal = model.multimodal; if (typeof model.features?.tools === 'boolean') descriptor.toolCalling = model.features.tools; + if (isPositiveFiniteNumber(model.max_input_tokens)) descriptor.maxInputTokens = model.max_input_tokens; + if (isPositiveFiniteNumber(model.max_output_tokens)) descriptor.maxOutputTokens = model.max_output_tokens; return descriptor; } diff --git a/src/cli/commands/proxy/connectors/vscode-models.ts b/src/cli/commands/proxy/connectors/vscode-models.ts index 6c7c45020..7b67139a0 100644 --- a/src/cli/commands/proxy/connectors/vscode-models.ts +++ b/src/cli/commands/proxy/connectors/vscode-models.ts @@ -45,6 +45,12 @@ const MESSAGE_AUTH_HEADERS = { Authorization: 'Bearer ${apiKey}', } as const; +/** + * Token limits here are fallbacks: they apply only when the tenant catalog + * does not report a limit for the model (see {@link resolveVsCodeTokenLimits}). + * The table's `maxInputTokens` is a prompt budget (window minus output), while + * the catalog value is not, hence the catalog input has the output subtracted. + */ export const VS_CODE_CAPABILITY_TABLE: readonly VsCodeCapabilityEntry[] = [ { family: 'claude-sonnet-4-5', @@ -377,3 +383,30 @@ export function buildDefaultVsCodeCapability(descriptor: TenantModelDescriptor): maxOutputTokens: DEFAULT_MAX_OUTPUT_TOKENS, }; } + +function pickTokenLimit(catalogValue: number | undefined, fallback: number): number { + return typeof catalogValue === 'number' && Number.isFinite(catalogValue) && catalogValue > 0 + ? catalogValue + : fallback; +} + +/** + * Resolve the token limits for a model entry. The output limit is the tenant + * catalog value when it is a finite number > 0, else the capability entry + * (table entry or default). The API input value is treated as the whole + * context window, so the resolved output limit is subtracted from it to make + * input plus output fit. When the catalog input is missing, or the + * subtraction is not positive, the entry's `maxInputTokens` is used. + */ +export function resolveVsCodeTokenLimits( + entry: VsCodeCapabilityEntry, + descriptor: TenantModelDescriptor +): { maxInputTokens: number; maxOutputTokens: number } { + const maxOutputTokens = pickTokenLimit(descriptor.maxOutputTokens, entry.maxOutputTokens); + const catalogInput = pickTokenLimit(descriptor.maxInputTokens, 0); + const promptBudget = catalogInput - maxOutputTokens; + return { + maxInputTokens: promptBudget > 0 ? promptBudget : entry.maxInputTokens, + maxOutputTokens, + }; +} diff --git a/src/cli/commands/proxy/connectors/vscode.ts b/src/cli/commands/proxy/connectors/vscode.ts index 998eab137..f9a3e13f6 100644 --- a/src/cli/commands/proxy/connectors/vscode.ts +++ b/src/cli/commands/proxy/connectors/vscode.ts @@ -3,10 +3,11 @@ import { mkdir, readFile, rename, stat, unlink, writeFile } from 'node:fs/promis import { homedir } from 'node:os'; import { dirname, join } from 'node:path'; import { ConfigurationError } from '@/utils/errors.js'; -import { fetchTenantModelDescriptors } from './tenant-catalog.js'; +import { fetchTenantModelDescriptors, type TenantModelDescriptor } from './tenant-catalog.js'; import { buildDefaultVsCodeCapability, findVsCodeCapabilityEntry, + resolveVsCodeTokenLimits, type VsCodeApiType, type VsCodeCapabilityEntry, type VsCodeReasoningEffort, @@ -110,12 +111,13 @@ function getApiPath(apiType: VsCodeApiType): string { function buildManagedModel( entry: VsCodeCapabilityEntry, - tenantId: string, + descriptor: TenantModelDescriptor, name: string, proxyUrl: string ): VsCodeManagedModel { + const { maxInputTokens, maxOutputTokens } = resolveVsCodeTokenLimits(entry, descriptor); const model: VsCodeManagedModel = { - id: tenantId, + id: descriptor.id, name, url: new URL(getApiPath(entry.apiType), proxyUrl).toString(), apiType: entry.apiType, @@ -123,8 +125,8 @@ function buildManagedModel( vision: entry.vision, streaming: true, thinking: entry.thinking, - maxInputTokens: entry.maxInputTokens, - maxOutputTokens: entry.maxOutputTokens, + maxInputTokens, + maxOutputTokens, }; if (entry.adaptiveThinking) model.adaptiveThinking = true; @@ -167,7 +169,7 @@ async function resolveManagedModels( const known = findVsCodeCapabilityEntry(descriptor.id); const entry = known ?? buildDefaultVsCodeCapability(descriptor); const name = known ? descriptor.id : (descriptor.label?.trim() || descriptor.id); - models.push(buildManagedModel(entry, descriptor.id, name, proxyUrl)); + models.push(buildManagedModel(entry, descriptor, name, proxyUrl)); } if (models.length === 0) { throw new ConfigurationError( diff --git a/src/providers/plugins/sso/sso.http-client.ts b/src/providers/plugins/sso/sso.http-client.ts index 83bff1136..9fc615b4d 100644 --- a/src/providers/plugins/sso/sso.http-client.ts +++ b/src/providers/plugins/sso/sso.http-client.ts @@ -194,10 +194,16 @@ export interface LlmModel { }; forbidden_for_web?: boolean; /** - * The model's maximum input context window in tokens (LiteLLM `model_info.max_input_tokens`). + * The model's maximum input tokens (LiteLLM `model_info.max_input_tokens`). This is the whole + * context window for some models and only the prompt budget for others, so callers must not + * assume it fits alongside the output limit. * Absent on routers and on catalogs served from static config rather than the LiteLLM proxy. */ max_input_tokens?: number; + /** + * The model's maximum output tokens (LiteLLM `model_info.max_output_tokens`). + */ + max_output_tokens?: number; /** * Present (and `true`) on a Switchyard-generated virtual router entry (`LlmRouterOption` in * the backend's `Union[LLMModel, LlmRouterOption]` response) — a `base_name` that itself