Skip to content

Commit c453972

Browse files
committed
fix(cli): stop storing Codex model_provider as the model
`session_meta` carries `model_provider` (the API provider id — `openai`, or whatever name a proxy gives itself) and no model at all; the model lives in `turn_context.model`. Reading the former into `model` is why `openai`, `crs` and `custom` showed up in the model leaderboard and priced at $0. The model is now seeded by scanning ahead for the first `turn_context`, so usage lines that precede it are still attributed. Two more names the pricing catalogue cannot resolve are normalized here: `codex-auto-review` (a label Codex stamps on automatic review turns) resolves to the review model shipping on that date, ported from ccusage's codex-auto-review-fallbacks.json, and a proxy-appended effort parenthetical (`gpt-5.5(xhigh)`) is stripped. Bumps BACKFILL_STATE_SCHEMA_VERSION to 8 so the next sync re-parses existing rollouts and rewrites the affected rows. Claude-Session: https://claude.ai/code/session_019u3RKiNJY1sRknveHJy9iJ
1 parent 9f76377 commit c453972

3 files changed

Lines changed: 180 additions & 10 deletions

File tree

‎packages/cli/src/adapters/codex.ts‎

Lines changed: 80 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -48,7 +48,9 @@ async function parseCodexSessionFile(
4848
let sessionId = sessionIdFromFilePath(filePath, 'codex')
4949
let cwd: string | undefined
5050
let project: string | undefined
51-
let model: string | undefined
51+
// `session_meta` has no model — only `model_provider` — so seed from the
52+
// first `turn_context` up front. See firstCodexTurnContextModel.
53+
let model: string | undefined = firstCodexTurnContextModel(lines)
5254
// Headless `codex exec` files have no session_meta; emit a synthetic
5355
// session.started on the first usage line so rollups still get a boundary.
5456
let sessionStartEmitted = false
@@ -149,7 +151,8 @@ async function parseCodexSessionFile(
149151
sessionId = stringField(payload, 'id') || sessionId
150152
cwd = stringField(payload, 'cwd') || cwd
151153
project = cwd ? path.basename(cwd) : project
152-
model = stringField(payload, 'model_provider') || model
154+
// Deliberately NOT reading `model_provider` here — see
155+
// firstCodexTurnContextModel; `model` was seeded from it above.
153156
events.push(withBackfillRefs({
154157
schemaVersion: AGENT_TIME_SCHEMA_VERSION,
155158
ts,
@@ -171,7 +174,7 @@ async function parseCodexSessionFile(
171174
currentTurnId = stringField(payload, 'turn_id') || currentTurnId
172175
cwd = stringField(payload, 'cwd') || cwd
173176
project = cwd ? path.basename(cwd) : project
174-
model = stringField(payload, 'model') || model
177+
model = normalizeCodexModel(stringField(payload, 'model'), ts) || model
175178
// Codex hasn't shipped service_tier inside turn_context yet, but the field
176179
// is the natural per-turn location and the upstream protocol allows it.
177180
// Honor it when present so future Codex builds get accurate fast/priority
@@ -189,12 +192,9 @@ async function parseCodexSessionFile(
189192
if (topType === 'turn.completed' || topType === 'result' || topType === undefined) {
190193
const usage = headlessCodexUsage(raw)
191194
if (usage) {
192-
const parsedModel = headlessCodexModel(raw)
193-
if (parsedModel) {
194-
model = parsedModel
195-
}
196-
const eventModel = parsedModel || model || 'gpt-5'
197195
const headlessTs = headlessCodexTimestamp(raw) || ts
196+
model = normalizeCodexModel(headlessCodexModel(raw), headlessTs) || model
197+
const eventModel = model || 'gpt-5'
198198
if (!sessionStartEmitted) {
199199
pushEvent(baseCodexEvent({
200200
ts: headlessTs,
@@ -566,6 +566,78 @@ async function parseCodexSessionFile(
566566

567567
// ── Codex-specific helpers ──
568568

569+
// Codex stamps `codex-auto-review` as the model on its automatic code-review
570+
// turns. It is a label, not a model — the tokens are billed against whichever
571+
// review model Codex shipped on that date. Ported from ccusage's
572+
// codex_log_model_fallback + codex-auto-review-fallbacks.json
573+
// (rust/crates/ccusage/src/adapter/codex/parser.rs). Newest first; keep in
574+
// sync when ccusage refreshes its snapshot. Note the table stops at gpt-5.5,
575+
// so any review turn after 2026-04-23 resolves to it — upstream has not
576+
// published a gpt-5.6-era row. That is not cost-neutral (gpt-5.6-sol carries
577+
// an explicit cache-creation rate gpt-5.5 does not), just the best available
578+
// evidence about which model actually ran.
579+
const CODEX_AUTO_REVIEW_MODEL = 'codex-auto-review'
580+
const CODEX_AUTO_REVIEW_FALLBACKS: ReadonlyArray<{ releasedOn: string, model: string }> = [
581+
{ releasedOn: '2026-04-23', model: 'gpt-5.5' },
582+
{ releasedOn: '2026-03-05', model: 'gpt-5.4' },
583+
{ releasedOn: '2026-02-05', model: 'gpt-5.3-codex' },
584+
{ releasedOn: '2025-12-11', model: 'gpt-5.2-codex' },
585+
{ releasedOn: '2025-11-13', model: 'gpt-5.1-codex' },
586+
{ releasedOn: '2025-09-15', model: 'gpt-5-codex' },
587+
{ releasedOn: '2025-08-07', model: 'gpt-5' },
588+
]
589+
590+
// Turn a raw Codex model string into a name the backend pricing catalogue can
591+
// resolve:
592+
// - `gpt-5.5(xhigh)` / `gpt-5.4 (high)` — some third-party Codex proxies
593+
// append the reasoning effort. It is not part of any catalogue id and
594+
// pricing never varies by effort, so drop it.
595+
// - `codex-auto-review` — resolve to the review model shipping at `ts`.
596+
function normalizeCodexModel(model: string | undefined, ts: string | undefined): string | undefined {
597+
if (!model) {
598+
return model
599+
}
600+
const cleaned = model.replace(/\s*\([^)]*\)\s*$/, '').trim() || model
601+
if (cleaned !== CODEX_AUTO_REVIEW_MODEL) {
602+
return cleaned
603+
}
604+
const date = ts?.slice(0, 10)
605+
const fallback = date && /^\d{4}-\d{2}-\d{2}$/.test(date)
606+
? CODEX_AUTO_REVIEW_FALLBACKS.find(entry => date >= entry.releasedOn)
607+
: undefined
608+
// Pre-dating the whole table (or an unparseable ts) means the oldest
609+
// release is the best guess — same default ccusage uses.
610+
return fallback?.model ?? 'gpt-5'
611+
}
612+
613+
// `session_meta` carries `model_provider` (an API provider id — `openai`, or
614+
// whatever name a proxy gives itself), never a model; the model lives in
615+
// `turn_context.model`. Scan ahead for the first one so events emitted before
616+
// the first turn_context (session.started, plus any usage line that precedes
617+
// it) still carry the real model. Reading `model_provider` into `model` is
618+
// how `openai` / `crs` / `custom` used to reach the model leaderboard.
619+
function firstCodexTurnContextModel(lines: string[]): string | undefined {
620+
for (const line of lines) {
621+
// Cheap reject first — re-parsing every line of a large rollout is not
622+
// worth it when turn_context normally sits within the first few lines.
623+
if (!line.includes('"turn_context"')) {
624+
continue
625+
}
626+
const raw = parseJsonLine(line)
627+
if (!raw || stringField(raw, 'type') !== 'turn_context') {
628+
continue
629+
}
630+
const model = normalizeCodexModel(
631+
stringField(objectField(raw, 'payload'), 'model'),
632+
timestampFrom(raw.timestamp),
633+
)
634+
if (model) {
635+
return model
636+
}
637+
}
638+
return undefined
639+
}
640+
569641
function rewriteCodexModelForTier(model: string | undefined, serviceTier: string | undefined): string | undefined {
570642
if (!model) {
571643
return model

‎packages/cli/src/lib/types.ts‎

Lines changed: 6 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -37,13 +37,17 @@ export interface BackfillSourceFile {
3737
// and the pi/gemini edge cases, and the v6 Codex UUIDv7 creation-anchor
3838
// that skips copied branch/goal rollout history, and the v7 realpath
3939
// canonicalization of source file paths so symlinked agent homes stop
40-
// producing duplicate rollup identities). The CLI compares the
40+
// producing duplicate rollup identities, and the v8 Codex model-name
41+
// fixes — `session_meta.model_provider` is no longer stored as the model,
42+
// the model is seeded from the first `turn_context`, `codex-auto-review`
43+
// resolves to the review model shipping on that date, and a proxy's
44+
// effort parenthetical is stripped). The CLI compares the
4145
// constant against the on-disk schema; on a mismatch it drops every
4246
// watermark so the next sync silently re-parses all jsonl from scratch
4347
// and upserts the rebuilt rollups (`replace: true` is already set) — no
4448
// purge, nothing deleted. Users get the fix transparently the next time
4549
// their agent runs.
46-
export const BACKFILL_STATE_SCHEMA_VERSION = 7
50+
export const BACKFILL_STATE_SCHEMA_VERSION = 8
4751

4852
export interface BackfillIncrementalState {
4953
version: typeof BACKFILL_STATE_SCHEMA_VERSION

‎packages/cli/test/codex.test.ts‎

Lines changed: 94 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -549,3 +549,97 @@ test('the same-second thread_spawn replay skip still works in a UUIDv7-named fil
549549
assert.equal(usages.length, 1)
550550
assert.equal(usages[0].metrics?.tokensInput, 100)
551551
})
552+
553+
// ── model naming ──
554+
//
555+
// The model that gets stored is the pricing key the backend looks up, so a
556+
// wrong name is a silently-$0 (or silently-mispriced) row rather than a
557+
// visible error.
558+
559+
// Token numbers are irrelevant to these tests — only the model stamped on
560+
// the resulting usage event is.
561+
function tokenCount(timestamp: string): unknown {
562+
return {
563+
timestamp,
564+
type: 'event_msg',
565+
payload: { type: 'token_count', info: { last_token_usage: { input_tokens: 100, cached_input_tokens: 10, output_tokens: 50, total_tokens: 150 } } },
566+
}
567+
}
568+
569+
test('session_meta.model_provider is never used as the model', async () => {
570+
// `model_provider` is the API provider id — `openai` for the real thing,
571+
// or whatever a third-party proxy calls itself. Reading it into `model`
572+
// is how `openai` / `crs` / `custom` used to reach the model leaderboard.
573+
const events = await parse([
574+
{ timestamp: '2026-01-02T00:00:00.000Z', type: 'session_meta', payload: { id: 'session', cwd: '/w', model_provider: 'openai' } },
575+
{ timestamp: '2026-01-02T00:00:01.000Z', type: 'turn_context', payload: { model: 'gpt-5.6-sol' } },
576+
tokenCount('2026-01-02T00:00:02.000Z'),
577+
])
578+
579+
assert.equal(events.every(event => event.model !== 'openai'), true)
580+
assert.equal(usageEvents(events)[0].model, 'gpt-5.6-sol')
581+
})
582+
583+
test('usage recorded before the first turn_context still carries the real model', async () => {
584+
// The model is seeded by scanning ahead for the first turn_context, so a
585+
// token_count that lands before it is not left model-less (or, previously,
586+
// stamped with the provider id).
587+
const events = await parse([
588+
{ timestamp: '2026-01-02T00:00:00.000Z', type: 'session_meta', payload: { id: 'session', cwd: '/w', model_provider: 'openai' } },
589+
tokenCount('2026-01-02T00:00:01.000Z'),
590+
{ timestamp: '2026-01-02T00:00:02.000Z', type: 'turn_context', payload: { model: 'gpt-5.6-sol' } },
591+
])
592+
593+
assert.equal(usageEvents(events)[0].model, 'gpt-5.6-sol')
594+
assert.equal(events.find(event => event.type === 'session.started')?.model, 'gpt-5.6-sol')
595+
})
596+
597+
test('a rollout with no turn_context leaves the model unset rather than guessing', async () => {
598+
const events = await parse([
599+
{ timestamp: '2026-01-02T00:00:00.000Z', type: 'session_meta', payload: { id: 'session', cwd: '/w', model_provider: 'crs' } },
600+
tokenCount('2026-01-02T00:00:01.000Z'),
601+
])
602+
603+
assert.equal(usageEvents(events)[0].model, undefined)
604+
})
605+
606+
test('the reasoning-effort parenthetical some proxies append is stripped', async () => {
607+
// `gpt-5.5(xhigh)` is not a catalogue id, and pricing does not vary by
608+
// effort — the effort is already carried by other fields.
609+
const usages = usageEvents(await parse([
610+
{ timestamp: '2026-01-02T00:00:00.000Z', type: 'turn_context', payload: { model: 'gpt-5.5(xhigh)' } },
611+
tokenCount('2026-01-02T00:00:01.000Z'),
612+
]))
613+
614+
assert.equal(usages[0].model, 'gpt-5.5')
615+
})
616+
617+
test('parity: ccusage codex_log_model_fallback — codex-auto-review resolves by date', async () => {
618+
// Codex stamps `codex-auto-review` on its automatic review turns; the
619+
// tokens bill against whichever review model shipped on that date.
620+
// Mirrors ccusage's codex-auto-review-fallbacks.json snapshot.
621+
const cases: Array<[string, string]> = [
622+
['2026-05-01T00:00:00.000Z', 'gpt-5.5'],
623+
['2026-04-23T00:00:00.000Z', 'gpt-5.5'],
624+
['2026-04-22T00:00:00.000Z', 'gpt-5.4'],
625+
['2026-02-10T00:00:00.000Z', 'gpt-5.3-codex'],
626+
['2025-09-20T00:00:00.000Z', 'gpt-5-codex'],
627+
['2025-01-01T00:00:00.000Z', 'gpt-5'],
628+
]
629+
for (const [ts, expected] of cases) {
630+
const usages = usageEvents(await parse([
631+
{ timestamp: ts, type: 'turn_context', payload: { model: 'codex-auto-review' } },
632+
tokenCount(ts),
633+
]))
634+
assert.equal(usages[0].model, expected, `${ts} → ${expected}`)
635+
}
636+
})
637+
638+
test('the fast tier still suffixes the resolved model, not the raw label', async () => {
639+
const usages = usageEvents(await parse([
640+
{ timestamp: '2026-05-01T00:00:00.000Z', type: 'turn_context', payload: { model: 'codex-auto-review', service_tier: 'priority' } },
641+
tokenCount('2026-05-01T00:00:01.000Z'),
642+
]))
643+
644+
assert.equal(usages[0].model, 'gpt-5.5-fast')
645+
})

0 commit comments

Comments
 (0)