diff --git a/API.md b/API.md index 3dc1e6c..194d9dc 100644 --- a/API.md +++ b/API.md @@ -97,6 +97,8 @@ import { isCodexUserPromptLine, isCodexStatusLine, isCodexIntermediateChromeLine, detectCodexApproval, isApprovalOverlayVisible, detectCodexTrustDialog, CODEX_TRUST_DIALOG_ACCEPT_KEYS, + CODEX_TRUST_DIALOG_DECLINE_KEYS, CODEX_TRUST_DIALOG_FOLDER_ACCESS_ACCEPT_KEYS, + CODEX_TRUST_DIALOG_FOLDER_ACCESS_DECLINE_KEYS, diffLines, // Transcript isCodexConversationEntry, isCodexResponseItem, isCodexEventMsg, @@ -220,7 +222,7 @@ shared verbatim with `claude-code-headless` (see the header comment in | Transcript | `~/.claude/projects//.jsonl` (per-cwd) | `~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl` (date-bucketed globally) | | Assistant marker | `⏺` | `•` (or older `◦`) | | User marker | `❯` | `›` | -| Trust prompt | "Accessing workspace" | "Do you trust the contents of this directory" | +| Trust prompt | "Accessing workspace" | "Do you trust the contents of this directory" (≤ 0.149); "Folder access" / "Trust this folder?" (0.156+) | | Proxy | mitmproxy TLS interceptor | plain HTTP server behind `openai_base_url` | --- @@ -421,9 +423,13 @@ also carries `ts: number` (epoch ms). Notes on the `trust_dialog` action callbacks: -- `accept()` writes `CODEX_TRUST_DIALOG_ACCEPT_KEYS` (`'\r'`) — - confirms the pre-selected "Yes, continue". -- `reject()` writes `'2\r'` — selects "No, quit". +- `accept()` writes the matched layout's `acceptKeys`: `'1'` on the legacy + layout (selects "Yes, continue" at once); `'1\r'` on the 0.156+ layout, + where `1` only moves the highlight to option 1 and Enter confirms it. +- `reject()` writes the layout's `declineKeys`, `'2'` in both layouts. It + selects option 2 at once, with no trailing Enter to leak into the next + screen. On 0.156+ option 2 is "Quit", or "Back to Agent Command Center" + when Codex is connected to its background server. The `event` flat surface only emits a `{ type: 'trust_dialog', … }` member when the dialog becomes **visible**; the simple `trust-dialog` @@ -1138,7 +1144,7 @@ on-screen. | Field | Type | Description | | --- | --- | --- | | `state` | `CodexTrustDialogState` | The parsed trust dialog (§7.3). | -| `actions` | `ConditionAction[]` | `accept` (Trust folder, writes `'\r'`), `reject` (Quit, writes `'2\r'`). | +| `actions` | `ConditionAction[]` | `accept` and `reject`, writing the state's `acceptKeys` / `declineKeys` (above). Legacy labels are "Trust folder" / "Quit"; 0.156+ labels are the option texts painted on screen. | **`CodexApprovalCondition`** — `kind: 'codex.approval'` @@ -1301,23 +1307,27 @@ title matching, returns `boolean`. detectCodexTrustDialog(screen: string): CodexTrustDialogState ``` -Detects Codex's first-launch-in-a-new-directory trust dialog. **All** -required markers must be present (conservative — avoids -false-positiving on assistant text that mentions "trust"): -`Do you trust the contents of this directory`, `Yes, continue`, -`No, quit`. +Detects Codex's first-launch trust dialog in either upstream layout, **structurally**: the dialog's key hint must be the last painted row, with the adjacent option pair above it and the dialog's title line above that. A transcript that quotes the dialog, even verbatim, always has the live composer below it and never matches. + +- **Legacy (≤ 0.149.1):** `> You are in `, `1. Yes, continue` / `2. No, quit`, then `Press enter to continue`. +- **0.156+:** `Folder access` and the path, `1. Trust and continue` (or `Open restricted` / `Open existing task`), then `2. Quit` (or `Back to Agent Command Center`), then `enter continue · esc quit|back`. One wrap of the hint is accepted. `CodexTrustDialogState`: | Field | Type | Description | | --- | --- | --- | | `visible` | `boolean` | | -| `workspace?` | `string` | The directory Codex asks to trust, parsed from a `> You are in ` line. | -| `options?` | `Array<{ key: string; label: string }>` | The two options, hardcoded `{ '1', 'Yes, continue' }` / `{ '2', 'No, quit' }`. | +| `workspace?` | `string` | The folder the dialog names. Hard-wrapped path rows are joined. | +| `trustTarget?` | `string` | 0.156+ only: the Git repository root that trust applies to, when Codex is in a subdirectory. | +| `options?` | `Array<{ key: string; label: string }>` | The two options, labels as painted. | +| `layout?` | `'you-are-in' \| 'folder-access'` | Which layout matched. | +| `acceptKeys?` / `declineKeys?` | `string` | The bytes that choose option 1 / option 2 on that layout. | + +Constants: +- `CODEX_TRUST_DIALOG_ACCEPT_KEYS` = `'1'` and `CODEX_TRUST_DIALOG_DECLINE_KEYS` = `'2'` (legacy layout). +- `CODEX_TRUST_DIALOG_FOLDER_ACCESS_ACCEPT_KEYS` = `'1\r'` and `CODEX_TRUST_DIALOG_FOLDER_ACCESS_DECLINE_KEYS` = `'2'` (0.156+). -`CODEX_TRUST_DIALOG_ACCEPT_KEYS` = `'\r'` — confirms the pre-selected -"Yes, continue". (Reject is `'2\r'`, not exported as a constant — -see the trust-dialog condition's `reject` action.) +Prefer the state's `acceptKeys` / `declineKeys` over the constants. ### 7.4 Line diff — `LineDiff.ts` @@ -1601,7 +1611,7 @@ static ResponsesProxy.create(options?: { | --- | --- | --- | --- | | `authMode` | `CodexAuthMode` (`'apikey' \| 'chatgpt'`) | auto-detected | Which auth path Codex uses. Auto-detection reads `~/.codex/auth.json`; absent → `'apikey'`. | | `upstreamBaseUrl` | `string` | derived from `authMode` | The real upstream. Defaults: `apikey` → `https://api.openai.com/v1`, `chatgpt` → `https://chatgpt.com/backend-api/codex`. | -| `eventsFile` | `string` | unset | If set, every emitted `event` is also written as a JSON line to this file (forensic mirror; `Buffer` payloads are inlined as `{ _buffer_b64 }`. Dumps written before agent-code#372's fix hold `{"type":"Buffer","data":[…]}` byte arrays instead, so a reader of older files must accept both). Written asynchronously through one ordered stream, never on the emitting call. When a line would push the total unwritten bytes (across the live and any rotated-out file) past 16 MiB, it is dropped and counted. A `{kind:'mirror-dropped', droppedEvents, droppedBytes}` line precedes the next written line, or is written at `stop()` if no line follows. Markers obey the same bounds: with a queue cap smaller than one marker (only ever a test setting), the marker is not written and the drops stay counted in the mirror's stats. Call `flushMirror()` to wait for the queue; `stop()` flushes and closes it. | +| `eventsFile` | `string` | unset | If set, every emitted `event` is also written as a JSON line to this file (forensic mirror; `Buffer` payloads are inlined as `{ _buffer_b64 }`. Dumps written before agent-code#372's fix hold `{"type":"Buffer","data":[…]}` byte arrays instead, so a reader of older files must accept both). Written asynchronously through one ordered stream, never on the emitting call. When a line would push the total unwritten bytes (across the live and any rotated-out file) past 16 MiB, it is dropped and counted. A `{kind:'mirror-dropped', droppedEvents, droppedBytes}` line precedes the next written line, or is written at `stop()` if no line follows. Markers obey the same bounds: with a queue cap smaller than one marker (only ever a test setting), the marker is not written and the drops stay counted in the mirror's stats. Also keeps the newest main-turn `responses*` request body in `latest-request-body.json` beside the file: one line, `{kind:'request-body-latest', requestId, endpoint, body_b64}` (raw on-wire bytes, zstd on current Codex), the name and kind of the Claude addon's sidecar. Requests with an output schema (`request_shape.has_output_schema`, e.g. title generation), subagent requests (`x-openai-subagent` other than `compact`), and bodiless requests never replace it. A newer body removes the old file at once. A write superseded by a newer body never lands, and a body over 16 MiB leaves no file, so the file is never an older prompt. Writes are asynchronous, encoded in 768 KiB slices, and keep at most one pending body. Call `flushMirror()` to wait for the queue and the sidecar; `stop()` flushes and closes them. | | `eventsFileMaxBytes` | `number` | 64 MiB | When the next line would pass this size, the file is renamed to `.1.jsonl` (replacing any earlier one) and a fresh file starts with `{kind:'mirror-rotated', rotatedBytes, rotations, droppedEvents, droppedBytes}`. Every file stays within the cap, markers included, so a run holds at most two files of this size. A line that cannot fit even in a fresh file is dropped and counted. A restart that appends to a file already at the cap rotates on its first event. | `create()` starts listening before resolving. The resolved instance diff --git a/SECURITY.md b/SECURITY.md index 7fb7255..bf1c763 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -42,6 +42,13 @@ While the proxy is in use: `Authorization` header** so a leaked bundle cannot expose bearer tokens — but it **does** record request bodies (your prompts, base64, capped at 2 MiB) and response bytes. Treat that file as sensitive. +- With `eventsFile` set, the proxy also keeps the newest main-turn Responses + request body (up to 16 MiB) in `latest-request-body.json` next to it, so a + debug bundle has the prompt even after the events tail has moved past it + (agent-code#1336). A body between 2 MiB and 16 MiB is persisted **only** + there: the mirror omits it. A crash between writing and renaming can also + leave a `latest-request-body.json...tmp` copy beside it. Treat all + of these as sensitive. The proxy is **opt-in**. If you don't construct it, no interception happens — Codex talks to the upstream directly and the package observes diff --git a/docs/plans/2026-09-27-latest-request-body.md b/docs/plans/2026-09-27-latest-request-body.md new file mode 100644 index 0000000..1e783cc --- /dev/null +++ b/docs/plans/2026-09-27-latest-request-body.md @@ -0,0 +1,58 @@ +# Keep the newest Responses request body beside the events file (agent-code#1336) + +## Evidence +- **What the bundle carries.** Agent Code's debug bundle includes the last 5 MiB of `proxy-events.jsonl` (plus `.1`). +- **Why the prompt falls out.** On Codex the response chunks are the bulk of that file. So after a long stream, or a run of `/v1/models` refreshes, the `request` event with its `body_b64` is no longer in the tail, and the bundle has no prompt. +- **Measured.** Review C of agent-code#1332 found the last inline body more than 5 MiB before EOF in 40 of 65 Codex files: 10 of 65 after the base64 fix (#53). +- **The Claude precedent.** Claude solved the same loss with `latest-request-body.json` (claude-code-headless#62, agent-code#1273). Agent Code's `readLatestRequestBody` already appends `/latest-request-body.json` for any provider. Codex never writes one. + +## Change +- **New module, `src/proxy/latestRequestBody.ts`:** `LatestRequestBodySidecar`. For each `responses*` request with a body, it replaces the sidecar with `{kind:'request-body-latest', requestId, endpoint, body_b64}`. + - **Ordered writes.** One ordered chain of async writes (temp file plus rename): no sync I/O on the main process, and an older body can never rename over a newer one. + - **Removal instead of staleness.** It removes the sidecar when the newest body is over 16 MiB (the Claude cap) or a write fails, so it never shows an older prompt as current. + - **Bodiless requests are skipped.** `/models` GETs never touch it. +- **Wiring in `ResponsesProxy`.** It is created with the mirror (same `eventsFile` opt-in) and recorded right after the `request` event. `flushMirror()` and `stop()` also wait for it. +- **Docs.** `API.md` and `SECURITY.md` document the file, including that it holds prompt text. + +## Invariant, and how it differs from Claude +Claude's sidecar holds the newest body that is NOT in its log, because that log omits bodies past a budget. Codex never omits bodies up to 2 MiB; the loss is in the bundle's tail, which the proxy cannot see. So this sidecar always holds the newest Responses body, even when the log also has it. The cost is at most one body appearing twice in a bundle. + +## Tests (`responsesProxy.latestBody.test.ts`, real requests through the proxy to a local upstream) +1. After two POSTs, the sidecar holds the second body, with its `requestId`, as one line. +2. A `/models` GET after a prompt leaves the prompt in place. +3. A body over 16 MiB removes an existing sidecar. + +All three are red with the wiring removed. + +## Then +The app bump carries this to Agent Code. Its reader's WHY comment, which describes only the Claude invariant, gets the Codex one. + +## Review a (round 1) +- **Stale "latest"**, after a crash between write and rename, a failed removal, or a read while a newer write was queued. Fixed with three rules (see the module header): + - `record()` unlinks the real-name file synchronously; + - a superseded write never renames: the generation check and `renameSync` run in one synchronous turn (steering q96: an awaited async rename after the check let `record(B)` unlink the file and A's late rename restore it); + - only the newest pending body is kept. + If the directory refuses the unlink, nothing can make the file current; that stays unreported, like every mirror failure. +- **Title generation replaced the main prompt.** Codex 0.157 title turns carry an output schema (`text.format`, `codex_output_schema`). `request_shape` gains `has_output_schema`, and such requests are skipped. A body whose shape cannot be read is still kept. +- **Main-process cost.** Base64 is encoded in 768 KiB slices, one per turn after a drain, instead of 21 MiB in one call. At most one body is pending. +- **SECURITY.md accuracy.** Bodies between 2 MiB and 16 MiB are persisted only in the sidecar. +- **Fresh sessions never reach the file (P1).** This is an app-side selection bug, not a package bug: the bundle asks for `resume-` while a fresh run lives under `shell-`. It is fixed in the app PR that bumps this package, where it is reachable and testable. +- **Tests:** compaction is recorded, the title turn is skipped, a newer record removes the file at once, a superseded write cannot resurrect an older body, and a multi-slice body round-trips. + - Mutations killed: the endpoint narrowing, the schema check, the generation check and the unlink. + +## Steering q96 +- **A late async rename restored an older body.** The first round checked the generation, then awaited an async `rename`. A `record(B)` in that gap unlinked the public file, and A's rename then restored A while B was still being written. +- **Fix: commit in one turn.** The commit is now `renameSync` in the same turn as the check. It is one metadata call; body bytes are still written asynchronously in slices. +- **Test: `latestRequestBody.commitFence.test.ts`.** It holds every async `fs/promises` rename at a gate and spies `renameSync`. It records B while A's commit is pending, releases A, and reads the public file before B commits: the file is absent or B, never A. It was red on `a54cfe7` (A restored). + +## Review b (round 1) +- **Subagent calls replaced the main prompt.** Codex tags every non-main Responses call with `x-openai-subagent` (codex-api `requests/headers.rs`: `review`, `compact`, `thread_spawn`, `memory_consolidation`, or a label). They are now skipped, except `compact`, which carries the main conversation. There are tests for `thread_spawn` and `review`, and for a kept `compact`. +- **16 MiB cap untested.** A body of exactly 16 MiB is now kept, so lowering the cap fails a test. +- **Crash-left temp copies.** They are now listed in `SECURITY.md`. +- **P1 (the app side), no package change.** Covered by agent-code#1399, plus a bump that follows #1366. This PR is "For", not "Fixes", #1336. + +## Review c +- **The endpoint filter was unpinned, and my "mutations killed" claim for it was false.** The `/models` test is bodiless, so the filter was never consulted. A new test sends bodied `/memories/trace_summarize` and `/alpha/search` POSTs after a main prompt, and the prompt must stay. Replacing the filter with `true` now fails it. +- **`has_output_schema` is a heuristic, not a title detector.** Output schemas are a per-turn option upstream. Agent Code sends none on main turns, and a structured main turn would leave the sidecar one turn behind. The comment now says so. +- **Crash-left temp files were cleaned only after a failure.** They are now also swept after a successful commit, with a test. +- **Subagent label list in the comment.** Corrected to upstream's (`collab_spawn`, `guardian`, …). The rule itself is label-agnostic. diff --git a/docs/plans/2026-09-27-prompt-input-profile-0157.md b/docs/plans/2026-09-27-prompt-input-profile-0157.md new file mode 100644 index 0000000..5a1cc00 --- /dev/null +++ b/docs/plans/2026-09-27-prompt-input-profile-0157.md @@ -0,0 +1,60 @@ +# Prompt-input profile for Codex 0.157.1 (#63) + +## Evidence +- `prepareCodex01491PromptInputProfile` refuses any CLI whose app-server `userAgent` is not exactly `0.149.1`. + - The result at 0.157.1, the version Agent Code runs daily, is `unsupported-cli`. + - `CodexHeadless` then gets no profile and `PromptInputEvidence` yields nothing. + - Fresh-rollout ownership survives only through the proxy identity path. +- The pin is deliberate. The profile is a recorded contract, not a version range. The fix is to record 0.157.1 against the same corpus and issue a profile only for a version whose recording matches. +- **First run of the unchanged recorder against 0.157.1 (2026-09-27):** + 1. **Codex never started.** 0.157 auto-starts a managed app-server daemon whose control socket lives under `CODEX_HOME`. The recorder's isolated home under macOS `$TMPDIR` makes the socket path longer than SUN_LEN, and Codex exits with "app server did not become ready … use --no-daemon". The spawned daemon also outlived its deleted home. Filed as #66. + - `--no-daemon` exists from 0.156.0: absent in the installed 0.150.1–0.155.1 standalone releases, present in 0.156.0–0.157.1. + 2. **The trust case waited forever** for the pre-0.156 dialog text. The 0.156+ dialog is different (#65). There, `1` only moves the highlight and Enter confirms. +- **Agent Code does not pass `--no-alt-screen`**, and 0.157 made the fullscreen transcript (alternate screen) the default. The recorder always passed `--no-alt-screen`, so its inline corpus alone does not describe the screen the app's panes show. + +## Change (recorder, this commit's scope) +- Pass `--no-daemon` from 0.156, gated by version because older CLIs reject unknown flags. +- `CODEX_INPUT_RECORD_ALT_SCREEN=1` records with the app's launch shape (no `--no-alt-screen`). +- **The trust case handles both dialogs.** On the 0.156+ dialog it writes `1`, then asserts after 1 s that the dialog is STILL up, then writes Enter. This live-verifies (in the isolated home) the keystroke contract that #65 took from upstream source. + +## Result (recorded 2026-09-27) +- **Corpus.** All 16 cases were recorded against codex-cli 0.157.1, inline and fullscreen. Rollout and request agree in every case. The only provider semantic that differs from 0.149.1 is the Vim-default composer opening in Insert (`vim-normal-default` submits `iabc`), and the issued profile forces Vim off anyway. +- **`config/read`.** The projection is identical to 0.149.1's. The per-tag audit of the config precedence code is in `codex-01571-config-source.json`: all nine claims hold at new coordinates. New in 0.157: thread config layers, which use the same precedence table. + +### Recorder fixes the recording needed +Each is gated to 0.156+ and explained where it is made: +- `--no-daemon` (SUN_LEN, #66). +- A seen model migration. +- Skill frontmatter. +- The trust case: `1` then Enter, asserting `1` alone leaves the dialog up. This is live verification for #65. +- Title side requests (`thread_title.rs`) are answered unheld and unrecorded. My first diagnosis called them retries; that was wrong, and the comment says so. +- The request that carries the prompt is matched, not the newest one (0.157 does re-send a held request). +- Frames and screens are windowed on the last painted row (0.157 paints short sessions from the top). +- The resize redraw is taken when the rows actually change. Fullscreen repaints on a frame timer after a quiet gap. +- The popup wait uses the popup's own hint. + +### Classifier fixes: real 0.157 surfaces the 0.149.1 classifier misread +Each is pinned by the recorded corpora: +- **The trust dialog's hint row has the idle-footer shape.** The dialog read as a composer drafting "1. Trust and continue". It is now matched as a bottom-row structural modal check, shared byte-for-byte with #67. +- **The skill popup moved above the composer**, with the hint `enter insert · esc close`. Read as an idle composer, Enter-to-insert would have produced evidence for a prompt Codex never sent. It is now `completion-popup`. +- **Fullscreen paints a two-row footer:** the status line, then `? for shortcuts` or `tab to queue message`. Only those exact rows are accepted; anything else is `unknown`. + +### Profile +Issued for a table of exact recorded versions (0.149.1, 0.157.1), never a range. Unrecorded versions still get `unsupported-cli`. + +### Fail-first +Against origin/main, both 0.157.1 suites fail at profile issuance. With only the classifier reverted, 6 tests fail. + +### Found and filed, not fixed here +- **#68:** 0.157.1 drops a `?` from a single-chunk typed draft in about 1 of 5 runs. That is the `sendPrompt` single-line path. The committed corpus is from intact runs, and the catalog says so. + +## Review round (a, b): popups above the composer +Both reviewers found the same critical gap. +- **What 0.157.1 paints.** It paints EVERY popup above the composer (`chat_composer.rs`, `popup_state.rs`). The slash-command and file popups have no hint row, the unified-mention popup has `enter/tab insert · esc close · …`, and a short skill popup omits its hint. +- **The failure.** A frame with a popup open read as an idle composer holding the draft, so Enter, which selects the popup item, produced evidence for a prompt Codex never sent. Reviewer b reproduced it with upstream's `slash_popup_footer_wide` snapshot: `/m` over `/memories`, giving false evidence of `/m`. +- **The fix is fail-closed on the draft.** Codex opens these popups from the draft itself (a leading `/`; an `@` or `$` token), so such a draft never yields prompt evidence. A real prompt starting with `/` or mentioning `$HOME` becomes a safe miss (proxy fallback), never a false prompt. The unified-mention hint is also recognised structurally. +- **Evidence.** + - Two new recorded cases on 0.156+: `slash-popup-enter-selects-command` (`/stat` + Enter dispatches `/status`) and `file-popup-enter-inserts-mention` (`@READ` + Enter inserts `README.md`). Neither submitted anything, and both corpora (inline and fullscreen) were re-recorded with them. + - Upstream's snapshot is a unit test. + - Removing the draft rule fails the slash cases in both corpora, and the snapshot test. +- **Also from review a.** The fullscreen `? for shortcuts` row is now asserted. The claim that the `config/read` projection is identical covers `effectiveInputProjection` only; the 0.149.1 fixture's extra `layerShapeEvidence` has no 0.157.1 counterpart. diff --git a/docs/plans/2026-09-27-trust-dialog-0157.md b/docs/plans/2026-09-27-trust-dialog-0157.md new file mode 100644 index 0000000..cbd754a --- /dev/null +++ b/docs/plans/2026-09-27-trust-dialog-0157.md @@ -0,0 +1,75 @@ +# Codex 0.157 trust dialog: new layout, and `1` no longer accepts (#65) + +## Evidence +- **Recording.** codex-cli 0.157.1 was recorded in a fresh untrusted folder (80×24, no keystroke sent), stored in `testing/fixtures/trust-dialog-0157/folder-access-back.json`. The dialog it paints: + ``` + Folder access + /private/tmp/…/untrusted-ABCD (hard-wrapped over 2 rows) + + Trust this folder? Codex can read, edit, and run files here, subject to your + … + › 1. Trust and continue + 2. Back to Agent Command Center + + enter continue · esc back + ``` + `detectCodexTrustDialog` returns `{visible:false}` for it. The parser anchors on `> You are in`, "Do you trust the contents of this directory", `1. Yes, continue` and `2. No, quit`, and none of these exist any more. +- **Upstream source**, `codex-rs/tui/src/onboarding/trust_directory.rs` at tag `rust-v0.157.1`, read in `vendor/codex-src`: + - **Option 1** is "Trust and continue". It becomes "Open restricted" when the folder is saved as untrusted, and "Open existing task" when an existing task is resumed while connected. + - **Option 2** is "Quit", or "Back to Agent Command Center" when connected to the background server (`TrustCancelAction::AgentsOverview`). The hint row ends "esc quit" or "esc back" to match. An earlier 0.157.1 run in a different folder showed "2. Quit", so both labels occur locally. + - **The keys changed.** `1`/`y` (`SELECT_FIRST`) now only MOVES THE HIGHLIGHT to option 1: "A terminal response fragment can start with `1`; trust always requires an explicit Enter confirmation". Only Enter (`CONFIRM`) confirms the highlighted row. `2`/`n`, `q`, Ctrl+C, Ctrl+D and Esc still act at once (`handle_quit`). + - **Other rows.** A Git subdirectory adds "Note: You’re in a subdirectory of a Git project. Trusting will apply to the repository root:" and a second path. A failed trust write adds an error paragraph between the options and the hint. Paths hard-wrap at any character, or are centre-truncated with `…`. +- **What breaks today:** + 1. The dialog is never detected, so no `codex.trust-dialog` condition appears. Readiness waits on a blocking screen nobody is told about. + 2. Answering with the current accept bytes (`'1'`) would only move the highlight and leave the dialog up. + 3. `CodexHeadless`'s legacy `trust_dialog` event still sends `'2\r'` to reject. The parser's own comment says that leaks an Enter into the next screen. + +## Change +- `detectCodexTrustDialog` recognises both layouts. + - **Legacy layout (≤ 0.149.1, still the accepted version):** unchanged. + - **Folder-access layout (0.157):** a line that is exactly `Folder access`. Below it, a `1.` row and then a `2.` row. Below those, the `enter continue · esc quit|back` hint. + - **Option labels are read from the screen**, and each must be one of upstream's known labels (option 1: trust / open restricted / open existing task; option 2: Quit / Back to Agent Command Center). Only real renders can match, and prose that quotes the rows still cannot. + - **Workspace:** the path rows under `Folder access`, with the hard wrap joined. When the Git note is present, `trustTarget` is the repository root. +- **The state carries its own keystrokes** (`acceptKeys`, `declineKeys`), because they depend on the layout the parser matched: + - Legacy accept stays `'1'`: it selects at once in ≤ 0.149.1, and an extra Enter would leak into the composer. + - 0.157 accept is `'1\r'`: `1` forces the highlight onto option 1, then Enter confirms it. The result is deterministic whatever row was highlighted, and it never confirms "Quit" by accident. + - Decline stays `'2'` in both layouts. + - The exported `CODEX_TRUST_DIALOG_*_KEYS` constants keep their legacy meaning, and new `CODEX_TRUST_DIALOG_FOLDER_ACCESS_*` constants are added. +- **The condition's actions** use the state's keystrokes and the on-screen labels, so the app shows "Back to Agent Command Center" rather than "Quit" when that is what the key does. Action ids (`accept` / `reject`) are unchanged. +- **The legacy `trust_dialog` event** in `CodexHeadless` uses the state's keystrokes too, which fixes the stray `'2\r'`. +- **The two other anchors** move to the shared detector: `ScreenParser.isTrustDialogVisible` (streaming-text suppression) and `Codex01491ComposerSurface.isKnownNonComposerModal`. That keeps the old anchor and adds the 0.157 one. + +## Tests +- Replay the recording through the real `HeadlessTerminal` (batched drain replay, as in `ComposerState.recorded.test.ts`). The frame must detect as visible, with the joined workspace, the on-screen labels and the 0.157 keystrokes. Red on main. +- The upstream 0.157.1 insta snapshots (git subdirectory, restricted, existing task, trust error, 40-column truncation) as plain frames, labelled as upstream test output, not local recordings. +- Negative cases: + - Prose quoting every row. + - Rows above the anchor. + - A missing hint row. + - An unknown option label. + - The legacy tests stay green. +- The condition-module test pins that actions follow the state's keystrokes and labels. + +## Out of scope +- The 0.157 "Cannot use the background server" screen (#66). +- The prompt-input profile (#63) and the write-stdin approval (#64). +- Answering the dialog live: that would write the user's `~/.codex/config.toml`. The keystroke semantics come from upstream source, which is stated as such. + +## Review round (3 reviewers) and live evidence +- **Copied-frame phantom (a, b).** A transcript quoting the dialog verbatim was detected as live, so a blocking dialog appeared that could be answered with keys. Detection is now anchored at the bottom: the key hint must be the last painted row (one wrap allowed), the nearest adjacent option pair sits above it, and the nearest `Folder access` above that. The dialog is the onboarding screen, painted before any chat widget, while a quoted copy always has the live composer below it. +- **Windows hint wrapped at narrow widths (a, b).** One wrap of the hint is accepted, and the composer-surface anchor tolerates it too. +- **Coverage gaps (b), all now tested:** + - the `CodexHeadless` event callbacks, driven through the public class with the recording; + - both option-2 condition labels; + - an unknown option-1 label; + - option adjacency; + - legacy streaming-text suppression through `ScreenParser`. +- **The keystrokes are now live-verified.** The prompt-input recorder (#63) ran codex-cli 0.157.1 on this dialog in an isolated `CODEX_HOME`: + - after `1` alone the dialog was still up 1 s later; + - the following Enter reached the composer. + + That confirms `'1\r'` beyond upstream source. It has been re-run several times, most recently on 2026-09-27. +- **Reviewer c (MERGE-READY with three follow-ups, all fixed here):** + - **F1: the legacy layout had the same copied-frame phantom** (pre-existing on main). The legacy detector is now anchored at the bottom the same way: "Press enter to continue" is the last row in rust-v0.149.1's trust_directory.rs and in the recorded 0.149.1 frame, and the Windows variant may wrap once. + - **F2: the composer-surface anchor was untested.** Writing that test exposed a real bug on this branch: unwrapped, the hint has the idle-footer shape, so the dialog was read as a composer whose draft is "1. Trust and continue". The anchor is now a bottom-row structural check (`TRUST_HINT_FOOTER`, shared byte-for-byte with #63's branch), pinned by `Codex01491ComposerSurface.test.ts`. The redundant window-based anchor is gone. + - **F3: `API.md` described stale callback bytes and the old substring detector.** It is rewritten for both layouts, and the new constants and the layout type are exported from the package entry. diff --git a/src/CodexHeadless.trustDialog.test.ts b/src/CodexHeadless.trustDialog.test.ts new file mode 100644 index 0000000..42d6660 --- /dev/null +++ b/src/CodexHeadless.trustDialog.test.ts @@ -0,0 +1,47 @@ +import type { IPty } from 'node-pty' +import { readFileSync } from 'node:fs' + +import { afterEach, expect, it } from 'vitest' + +import { CodexHeadless } from './CodexHeadless.js' + +// #67 review (a and b): the legacy `trust_dialog` event carries accept/reject +// callbacks that write to the PTY, and a mutation that made accept write the +// legacy '1' survived every parser test. On 0.156+ that '1' only moves the +// highlight and leaves Codex waiting on the dialog. This drives the public +// class with the recorded 0.157.1 dialog and checks the bytes it writes. +type Recording = { cols: number; rows: number; events: Array<{ t: number; dir: string; data?: string }> } +const recording = JSON.parse(readFileSync( + new URL('../testing/fixtures/trust-dialog-0157/folder-access-back.json', import.meta.url), + 'utf8', +)) as Recording + +const stops: Array<() => Promise> = [] +afterEach(async () => { for (const stop of stops.splice(0)) await stop() }) + +it('answers the recorded 0.157.1 dialog with 1+Enter to accept and 2 to decline', async () => { + const listeners = new Set<(data: string) => void>() + const written: string[] = [] + const pty = { + write: (data: string) => { written.push(data) }, + resize: () => undefined, + onData: (listener: (data: string) => void) => { listeners.add(listener); return { dispose: () => listeners.delete(listener) } }, + onExit: () => ({ dispose: () => undefined }), + } as unknown as IPty + const headless = new CodexHeadless({ pty, cwd: '/recorded/untrusted', cols: recording.cols, rows: recording.rows }) + stops.push(() => headless.stop()) + const events: Array<{ type: string; accept?: () => void; reject?: () => void }> = [] + headless.on('event', (event: { type: string }) => { if (event.type === 'trust_dialog') events.push(event) }) + // start() also acquires the rollout file; only the screen path is under + // test, so attach the terminal the way start() does. + ;(headless as unknown as { terminal: { attach(): void } }).terminal.attach() + for (const event of recording.events) if (event.dir === 'out') for (const listener of listeners) listener(event.data!) + + // Waits on the event itself (the completion signal), never on a wall clock; + // a detector that never fires fails on the test timeout. + while (events.length === 0) await new Promise(resolve => setTimeout(resolve, 5)) + const [trust] = events + trust!.accept!() + trust!.reject!() + expect(written).toEqual(['1\r', '2']) +}) diff --git a/src/CodexHeadless.ts b/src/CodexHeadless.ts index 0264e5b..7865f2f 100644 --- a/src/CodexHeadless.ts +++ b/src/CodexHeadless.ts @@ -46,6 +46,7 @@ import { detectCodexTrustDialog, type CodexTrustDialogState, CODEX_TRUST_DIALOG_ACCEPT_KEYS, + CODEX_TRUST_DIALOG_DECLINE_KEYS, } from './parsers/TrustDialogParser.js' import { makeEvaluator, @@ -108,7 +109,7 @@ import type { // Transcript: ~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl // (date-bucketed globally, not per-cwd) // Markers: • for assistant, › for user (not ⏺ and ❯) -// Trust: "Do you trust the contents" (not "Accessing workspace") +// Trust: "Do you trust the contents" (<=0.149) / "Folder access" (0.156+) // // The consumer owns the PTY. This class never spawns or kills processes. @@ -1049,8 +1050,12 @@ export class CodexHeadless extends EventEmitter { type: 'trust_dialog', ts: Date.now(), workspace: trust.workspace, - accept: () => this.write(CODEX_TRUST_DIALOG_ACCEPT_KEYS), - reject: () => this.write('2\r'), + // The keystrokes come from the layout the parser matched (#65): + // 0.156+ needs `1` + Enter to accept, where `1` alone only moves + // the highlight. Reject used to be a hard-coded '2\r', whose Enter + // leaked into whatever screen followed the dialog. + accept: () => this.write(trust.acceptKeys ?? CODEX_TRUST_DIALOG_ACCEPT_KEYS), + reject: () => this.write(trust.declineKeys ?? CODEX_TRUST_DIALOG_DECLINE_KEYS), }) } } diff --git a/src/conditions/trustDialog.ts b/src/conditions/trustDialog.ts index a41e801..e156fc3 100644 --- a/src/conditions/trustDialog.ts +++ b/src/conditions/trustDialog.ts @@ -1,6 +1,8 @@ import { CODEX_TRUST_DIALOG_ACCEPT_KEYS, CODEX_TRUST_DIALOG_DECLINE_KEYS, + CODEX_TRUST_DIALOG_FOLDER_ACCESS_ACCEPT_KEYS, + CODEX_TRUST_DIALOG_FOLDER_ACCESS_DECLINE_KEYS, type CodexTrustDialogState, } from '../parsers/TrustDialogParser.js' import { defineModule } from './core/contract.js' @@ -49,6 +51,28 @@ const TRUST_DIALOG_ACTIONS: readonly ConditionAction[] = [ { kind: 'pty', id: 'reject', label: 'Quit', data: CODEX_TRUST_DIALOG_DECLINE_KEYS }, ] +// #65: the template above is the LEGACY layout's actions, and it is only the +// fallback now. Codex 0.156+ paints a different dialog whose keystrokes differ +// (`1` merely moves the highlight there, so accept is `1` + Enter) and whose +// option labels vary: option 2 is "Back to Agent Command Center" when Codex +// is connected to its background server, and option 1 is "Open restricted" for +// a folder saved as untrusted. The parser reports both per frame, so the +// actions follow the screen that is actually up. A fixed "Quit" button that +// in fact returned to the overview, or a fixed "Trust folder" that in fact +// opened the folder restricted, would tell the user something false. +// +// The action ids stay `accept` / `reject`: those are what the app and the +// phone key on. Only `label` and `data` follow the state. +function trustDialogActions(state: CodexTrustDialogState): ConditionAction[] { + if (state.layout !== 'folder-access') return TRUST_DIALOG_ACTIONS.map((a) => ({ ...a })) + const label = (key: string, fallback: string) => + state.options?.find((option) => option.key === key)?.label ?? fallback + return [ + { kind: 'pty', id: 'accept', label: label('1', 'Trust and continue'), data: state.acceptKeys ?? CODEX_TRUST_DIALOG_FOLDER_ACCESS_ACCEPT_KEYS }, + { kind: 'pty', id: 'reject', label: label('2', 'Quit'), data: state.declineKeys ?? CODEX_TRUST_DIALOG_FOLDER_ACCESS_DECLINE_KEYS }, + ] +} + // trustDialogModule — the headless-module form of the trust-dialog condition. // // `detect` takes the WHOLE input bundle and reaches into `inputs.trustDialog`, @@ -71,7 +95,9 @@ export const trustDialogModule = defineModule< // isolation contract requires a mutated snapshot not to poison later ones. // `{ ...a }` is a sufficient clone because ConditionAction fields are all // primitives (no nested objects to share). - actions: () => TRUST_DIALOG_ACTIONS.map((a) => ({ ...a })), + // Each call builds new objects (see trustDialogActions), so the isolation + // contract above still holds. + actions: (state) => trustDialogActions(state), }) // Legacy builder, re-implemented on top of the module so any external importer diff --git a/src/index.ts b/src/index.ts index 409e10e..edc7df6 100644 --- a/src/index.ts +++ b/src/index.ts @@ -81,6 +81,9 @@ export { detectCodexTrustDialog, CODEX_TRUST_DIALOG_ACCEPT_KEYS, CODEX_TRUST_DIALOG_DECLINE_KEYS, + CODEX_TRUST_DIALOG_FOLDER_ACCESS_ACCEPT_KEYS, + CODEX_TRUST_DIALOG_FOLDER_ACCESS_DECLINE_KEYS, + type CodexTrustDialogLayout, type CodexTrustDialogState, } from './parsers/TrustDialogParser.js' diff --git a/src/parsers/ScreenParser.ts b/src/parsers/ScreenParser.ts index 03fe2a1..158957d 100644 --- a/src/parsers/ScreenParser.ts +++ b/src/parsers/ScreenParser.ts @@ -170,21 +170,22 @@ export function isCodexIntermediateChromeLine(line: string): boolean { return false } -// --- Trust dialog detection (inlined) --- - -const TRUST_DIALOG_MARKERS = [ - 'Do you trust the contents of this directory', - 'Yes, continue', - 'No, quit', -] - +// --- Trust dialog detection --- +// +// Delegates to the structural detector. This used to be an inlined copy that +// asked whether three legacy phrases appeared ANYWHERE on screen, which (a) +// blanked the streaming text whenever an assistant merely quoted the dialog, +// the false positive TrustDialogParser was rewritten to stop, and (b) never +// matched the 0.156+ `Folder access` layout, whose phrases are all different +// (#65). One detector means one answer to "is the dialog up". function isTrustDialogVisible(screen: string): boolean { - return TRUST_DIALOG_MARKERS.every(m => screen.includes(m)) + return detectCodexTrustDialog(screen).visible } // Approval detection lives in ApprovalParser.ts — import for the // streaming text suppression check. import { isApprovalOverlayVisible } from './ApprovalParser.js' +import { detectCodexTrustDialog } from './TrustDialogParser.js' // --- Resume picker detection --- diff --git a/src/parsers/TrustDialogParser.recorded.test.ts b/src/parsers/TrustDialogParser.recorded.test.ts new file mode 100644 index 0000000..7ced585 --- /dev/null +++ b/src/parsers/TrustDialogParser.recorded.test.ts @@ -0,0 +1,77 @@ +import type { IPty } from 'node-pty' +import { readFileSync } from 'node:fs' + +import { afterEach, expect, it } from 'vitest' + +import { trustDialogModule } from '../conditions/trustDialog.js' +import { HeadlessTerminal } from '../terminal/HeadlessTerminal.js' +import { extractCodexStreamingText } from './ScreenParser.js' +import { detectCodexTrustDialog } from './TrustDialogParser.js' + +// #65: a raw PTY recording of codex-cli 0.157.1 showing its trust dialog in a +// fresh untrusted folder (see the fixture's `source`; no key was ever sent). +// Replayed through the real terminal so the frame the parser reads is xterm's +// own parse of Codex's bytes: the cursor-addressed word placement, the +// highlighted row's full-width padding and the hard-wrapped path all come +// from upstream, not from a hand-written string. Every assertion below was red +// on main, where this dialog read as not visible. +type Recording = { cols: number; rows: number; events: Array<{ t: number; dir: string; data?: string }> } +const recording = JSON.parse(readFileSync( + new URL('../../testing/fixtures/trust-dialog-0157/folder-access-back.json', import.meta.url), + 'utf8', +)) as Recording + +const terminals: HeadlessTerminal[] = [] +afterEach(() => { for (const terminal of terminals.splice(0)) terminal.dispose() }) + +// Batched feed with a drain between batches, for the reasons written down in +// ComposerState.recorded.test.ts (#57): draining is the completion signal, and +// no wall-clock deadline decides when the frame is "done". +async function replay(): Promise { + const listeners = new Set<(data: string) => void>() + const pty = { + write: () => undefined, + resize: () => undefined, + onData: (listener: (data: string) => void) => { listeners.add(listener); return { dispose: () => listeners.delete(listener) } }, + onExit: () => ({ dispose: () => undefined }), + } as unknown as IPty + const terminal = new HeadlessTerminal({ pty, cols: recording.cols, rows: recording.rows, snapshotIntervalMs: 1 }) + terminals.push(terminal) + terminal.attach() + const chunks = recording.events.filter(event => event.dir === 'out').map(event => event.data!) + for (let start = 0; start < chunks.length; start += 50) { + for (const chunk of chunks.slice(start, start + 50)) for (const listener of listeners) listener(chunk) + while ((terminal as unknown as { pendingWrites: number }).pendingWrites !== 0) await new Promise(resolve => setImmediate(resolve)) + } + return terminal +} + +it('detects the recorded 0.157.1 Folder access dialog with its on-screen labels and keystrokes', async () => { + const screen = (await replay()).snapshotPlain() + expect(screen).toContain('Folder access') + + const state = detectCodexTrustDialog(screen) + expect(state).toEqual({ + visible: true, + // Painted over two rows (hard wrap at 76 columns); joined back into one. + workspace: '/private/tmp/claude-501/-Users-fixture-user-Desktop-Development-agent-code/d0000000-0000-0000-0000-000000000000/scratchpad/rec/untrusted-ABCD', + options: [ + { key: '1', label: 'Trust and continue' }, + { key: '2', label: 'Back to Agent Command Center' }, + ], + layout: 'folder-access', + acceptKeys: '1\r', + declineKeys: '2', + }) + + // The condition the app renders: ids unchanged, but the reject button says + // what the key really does here (back to the overview, Codex keeps running). + expect(trustDialogModule.actions(state)).toEqual([ + { kind: 'pty', id: 'accept', label: 'Trust and continue', data: '1\r' }, + { kind: 'pty', id: 'reject', label: 'Back to Agent Command Center', data: '2' }, + ]) + + // The streaming-text extractor must treat the dialog as a blocking screen, + // not as assistant output. + expect(extractCodexStreamingText(screen)).toBe('') +}) diff --git a/src/parsers/TrustDialogParser.test.ts b/src/parsers/TrustDialogParser.test.ts index 48e978e..213770d 100644 --- a/src/parsers/TrustDialogParser.test.ts +++ b/src/parsers/TrustDialogParser.test.ts @@ -1,8 +1,14 @@ +import { readFileSync } from 'node:fs' + import { describe, expect, it } from 'vitest' +import { trustDialogModule } from '../conditions/trustDialog.js' +import { extractCodexStreamingText } from './ScreenParser.js' import { CODEX_TRUST_DIALOG_ACCEPT_KEYS, CODEX_TRUST_DIALOG_DECLINE_KEYS, + CODEX_TRUST_DIALOG_FOLDER_ACCESS_ACCEPT_KEYS, + CODEX_TRUST_DIALOG_FOLDER_ACCESS_DECLINE_KEYS, detectCodexTrustDialog, } from './TrustDialogParser.js' @@ -88,6 +94,42 @@ describe('detectCodexTrustDialog', () => { expect(detectCodexTrustDialog(inverted).visible).toBe(false) }) + it('ignores a verbatim legacy dialog quoted above the live composer', () => { + // Review c of #67: the legacy layout had the same copied-frame phantom the + // 0.156+ layout was fixed for. Its hint is the last painted row too. + const quoted = [ + '• Here is the old Codex screen I captured:', + REAL_DIALOG, + '', + '› Ask Codex to do anything', + '', + ' gpt-5.4 medium fast · ~/project', + ].join('\n') + expect(detectCodexTrustDialog(quoted).visible).toBe(false) + expect(extractCodexStreamingText(quoted)).toContain('Yes, continue') + }) + + it('requires the legacy options to be adjacent rows directly under the question', () => { + const scattered = [ + '> You are in /tmp/x', + ' Do you trust the contents of this directory?', + '• unrelated assistant paragraph one', + '› 1. Yes, continue', + '• unrelated assistant paragraph two', + ' 2. No, quit', + ' Press enter to continue', + ].join('\n') + expect(detectCodexTrustDialog(scattered).visible).toBe(false) + }) + + it('reads the legacy Windows hint wrapped over two rows', () => { + const wrapped = REAL_DIALOG.replace( + ' Press enter to continue', + ' Press enter to continue and create a\n sandbox...', + ) + expect(detectCodexTrustDialog(wrapped).visible).toBe(true) + }) + it('returns not-visible for empty input', () => { expect(detectCodexTrustDialog('').visible).toBe(false) }) @@ -100,3 +142,168 @@ describe('detectCodexTrustDialog', () => { expect(CODEX_TRUST_DIALOG_DECLINE_KEYS).toBe('2') }) }) + +// --- 0.156+ `Folder access` layout (#65) --- +// +// Upstream's own insta snapshots at rust-v0.157.1, verbatim (see the fixture's +// `evidence`). They cover the variants one local folder cannot produce. The +// recorded local frame lives in TrustDialogParser.recorded.test.ts. +type UpstreamSnapshot = { name: string; frame: string } +const upstream = (JSON.parse(readFileSync( + new URL('../../testing/fixtures/trust-dialog-0157/upstream-snapshots-0157.1.json', import.meta.url), + 'utf8', +)) as { snapshots: UpstreamSnapshot[] }).snapshots +const frame = (name: string) => upstream.find(snapshot => snapshot.name === name)!.frame + +describe('detectCodexTrustDialog on the 0.156+ Folder access layout', () => { + it('reads every upstream variant, with labels exactly as painted', () => { + const expected: Record = { + renders_snapshot_for_git_repo: { workspace: '/workspace/project', labels: ['Trust and continue', 'Quit'] }, + renders_snapshot_for_remote_git_subdirectory: { workspace: '/srv/remote/project/nested', trustTarget: '/srv/remote/project', labels: ['Trust and continue', 'Back to Agent Command Center'] }, + renders_snapshot_for_trust_error: { workspace: '/workspace/project', labels: ['Trust and continue', 'Quit'] }, + renders_restricted_folder: { workspace: '/workspace/project', labels: ['Open restricted', 'Back to Agent Command Center'] }, + existing_untrusted_task: { workspace: '/workspace/project', labels: ['Open existing task', 'Back to Agent Command Center'] }, + // Hard wrap at an arbitrary character: "…/long-nested-folde" + "r". + folder_picker_restricted_40x24: { workspace: '/workspace/project/long-nested-folder', labels: ['Open restricted', 'Back to Agent Command Center'] }, + // No spacer rows at all: the paragraph opener ends the path block. + long_checkout_40x13: { workspace: 'workspace/…/repository', labels: ['Trust and continue', 'Quit'] }, + long_repository_root_40x17: { workspace: 'workspace/…/repository/checkout', trustTarget: 'workspace/…/repository', labels: ['Trust and continue', 'Quit'] }, + // Only the repository root fits, so the folder row is elided and the + // trust target stands in for the workspace. + only_repository_root_fits_40x16: { workspace: 'workspace/…/repository', trustTarget: 'workspace/…/repository', labels: ['Trust and continue', 'Quit'] }, + } + expect(upstream.map(snapshot => snapshot.name).sort()).toEqual(Object.keys(expected).sort()) + for (const [name, want] of Object.entries(expected)) { + const state = detectCodexTrustDialog(frame(name)) + expect({ name, state }).toEqual({ + name, + state: { + visible: true, + workspace: want.workspace, + ...(want.trustTarget !== undefined ? { trustTarget: want.trustTarget } : {}), + options: [{ key: '1', label: want.labels[0] }, { key: '2', label: want.labels[1] }], + layout: 'folder-access', + acceptKeys: '1\r', + declineKeys: '2', + }, + }) + } + }) + + it('ignores prose that quotes every row of the new layout', () => { + // The same class of phantom as PROSE_FALSE_POSITIVE: an assistant pasting + // the dialog inside a sentence or a code literal. No anchor is a whole line. + const prose = [ + '• The new dialog says "Folder access" and offers', + " const ROWS = ['1. Trust and continue', '2. Quit']", + ' with the hint enter continue · esc quit.', + ].join('\n') + expect(detectCodexTrustDialog(prose).visible).toBe(false) + }) + + it('ignores a verbatim copy of the dialog inside a transcript', () => { + // Review of #67 (a and b): a pasted upstream .snap or copied terminal frame + // satisfied every whole-line anchor and raised an answerable phantom. What + // a copy cannot fake is position: the live composer is always below it. + for (const name of ['renders_snapshot_for_remote_git_subdirectory', 'renders_restricted_folder']) { + const transcript = [ + '• Here is the Codex screen I captured:', + frame(name), + '', + '› Ask Codex to do anything', + '', + ' gpt-6-sol high · ~/project', + ].join('\n') + expect(detectCodexTrustDialog(transcript).visible).toBe(false) + // And the quoted frame stays visible as assistant text. + expect(extractCodexStreamingText(transcript)).toContain('Folder access') + } + }) + + it('reads the Windows sandbox hint wrapped over two rows at narrow widths', () => { + // Upstream wraps "enter continue and create sandbox · esc quit" (46 columns + // with the inset) below 46 columns. Built from the upstream git_repo frame, + // which carries that hint, re-wrapped the way a 40-column Paragraph does. + const wrapped = frame('renders_snapshot_for_git_repo').replace( + /^\s*enter continue and create sandbox · esc quit\s*$/m, + ' enter continue and create sandbox ·\n esc quit', + ) + expect(wrapped).toContain('sandbox ·\n esc quit') + expect(detectCodexTrustDialog(wrapped).visible).toBe(true) + }) + + it('rejects an option 1 label upstream cannot paint', () => { + expect(detectCodexTrustDialog(frame('renders_snapshot_for_git_repo').replace('1. Trust and continue', '1. Delete folder')).visible).toBe(false) + }) + + it('requires the two options to be adjacent rows', () => { + expect(detectCodexTrustDialog(frame('renders_snapshot_for_git_repo').replace(/(1\. Trust and continue[^\n]*)\n/, '$1\n\n')).visible).toBe(false) + }) + + it('requires the key hint below the options', () => { + expect(detectCodexTrustDialog(frame('renders_snapshot_for_git_repo').replace(/\n.*enter continue.*$/m, '')).visible).toBe(false) + }) + + it('requires the options below the Folder access anchor', () => { + const lines = frame('renders_snapshot_for_git_repo').split('\n') + const anchor = lines.findIndex(line => line.trim() === 'Folder access') + const moved = [...lines.slice(anchor + 1), lines[anchor]].join('\n') + expect(detectCodexTrustDialog(moved).visible).toBe(false) + }) + + it('rejects an option label upstream cannot paint', () => { + expect(detectCodexTrustDialog(frame('renders_snapshot_for_git_repo').replace('2. Quit', '2. Delete folder')).visible).toBe(false) + // With "esc back" the hint agrees with any non-Quit label, so only the + // label whitelist rejects this one (verification a of #67). + expect(detectCodexTrustDialog(frame('renders_restricted_folder').replace('2. Back to Agent Command Center', '2. Delete folder')).visible).toBe(false) + }) + + it('rejects a hint that contradicts option 2', () => { + // Upstream derives both from one TrustCancelAction, so "Quit" with + // "esc back" is not a real render. + expect(detectCodexTrustDialog(frame('renders_snapshot_for_git_repo').replace('esc quit', 'esc back')).visible).toBe(false) + }) +}) + +describe('keystrokes and condition actions per layout', () => { + it('keeps the legacy layout on the legacy keys and labels', () => { + const state = detectCodexTrustDialog(REAL_DIALOG) + expect(state.layout).toBe('you-are-in') + expect([state.acceptKeys, state.declineKeys]).toEqual(['1', '2']) + expect(trustDialogModule.actions(state)).toEqual([ + { kind: 'pty', id: 'accept', label: 'Trust folder', data: '1' }, + { kind: 'pty', id: 'reject', label: 'Quit', data: '2' }, + ]) + }) + + it('accepts the new layout with 1 then Enter, because 1 alone only moves the highlight there', () => { + // rust-v0.157.1 trust_directory.rs: SELECT_FIRST sets the highlight, + // CONFIRM (Enter) acts on it. Enter alone could confirm option 2. + expect(CODEX_TRUST_DIALOG_FOLDER_ACCESS_ACCEPT_KEYS).toBe('1\r') + expect(CODEX_TRUST_DIALOG_FOLDER_ACCESS_DECLINE_KEYS).toBe('2') + const state = detectCodexTrustDialog(frame('renders_restricted_folder')) + expect(trustDialogModule.actions(state)).toEqual([ + { kind: 'pty', id: 'accept', label: 'Open restricted', data: '1\r' }, + { kind: 'pty', id: 'reject', label: 'Back to Agent Command Center', data: '2' }, + ]) + }) + + it('labels option 2 Quit when that is what the screen says', () => { + // Both option-2 texts are live on the same binary; the Back variant alone + // would not catch a condition that always said "Back" (review of #67 b). + const state = detectCodexTrustDialog(frame('renders_snapshot_for_git_repo')) + expect(trustDialogModule.actions(state).map(action => action.label)).toEqual(['Trust and continue', 'Quit']) + }) + + it('keeps blanking the streaming text for the legacy dialog', () => { + // ScreenParser now delegates to this detector; pin the legacy side too. + expect(extractCodexStreamingText(REAL_DIALOG)).toBe('') + }) + + it('hands every caller its own action objects', () => { + const state = detectCodexTrustDialog(frame('renders_snapshot_for_git_repo')) + const first = trustDialogModule.actions(state) + first[0]!.label = 'mutated' + expect(trustDialogModule.actions(state)[0]!.label).toBe('Trust and continue') + }) +}) diff --git a/src/parsers/TrustDialogParser.ts b/src/parsers/TrustDialogParser.ts index 151bf9e..2cfe5e5 100644 --- a/src/parsers/TrustDialogParser.ts +++ b/src/parsers/TrustDialogParser.ts @@ -1,6 +1,8 @@ // Detect Codex's trust dialog from a screen snapshot. // -// Codex shows this on first launch in a new directory. Captured live from +// Two upstream layouts are recognised. The 0.156+ `Folder access` layout is +// documented at detectFolderAccessLayout below (#65). The legacy layout, still +// painted by the accepted 0.149.1, was captured live from // codex-cli 0.145.0 in a fresh temp dir (see // docs/decomposition/provider-condition-answering.md in agent-code): // @@ -17,12 +19,34 @@ export type CodexTrustDialogState = { /** True if Codex is currently showing the trust dialog. */ visible: boolean - /** The directory Codex is asking the user to trust. */ + /** + * The folder the dialog names: `> You are in ` (legacy) or the path + * under `Folder access` (0.157). When 0.157 elides that row for space, this + * falls back to `trustTarget`. + */ workspace?: string - /** The selectable options. */ + /** + * 0.157 only: the Git repository root that trust will actually apply to, + * shown under "Note: You’re in a subdirectory of a Git project". Absent when + * trust applies to `workspace` itself. + */ + trustTarget?: string + /** The selectable options, labels exactly as painted. */ options?: Array<{ key: string; label: string }> + /** Which upstream layout matched; decides the keystrokes below. */ + layout?: CodexTrustDialogLayout + /** Bytes that choose option 1 on THIS layout. See the constants below. */ + acceptKeys?: string + /** Bytes that choose option 2 on THIS layout. */ + declineKeys?: string } +/** + * `you-are-in`: codex-cli 0.145–0.149 (`> You are in …` / `1. Yes, continue`). + * `folder-access`: codex-cli 0.156+ (`Folder access` / `1. Trust and continue`). + */ +export type CodexTrustDialogLayout = 'you-are-in' | 'folder-access' + // STRUCTURAL anchoring, not substring presence. // // The previous implementation asked `screen.includes(marker)` for three @@ -68,22 +92,54 @@ const YOU_ARE_IN_RE = /^\s*>\s*You are in\s+(.+?)\s*$/ // keys, so either row may or may not be marked. const YES_ROW_RE = /^\s*[›>]?\s*1\.\s*Yes, continue\s*$/ const NO_ROW_RE = /^\s*[›>]?\s*2\.\s*No, quit\s*$/ +const LEGACY_HINT_RE = /^\s*Press enter to continue(?: and create a sandbox\.\.\.)?\s*$/ /** * Detect Codex's trust dialog from a plain-text screen snapshot. * - * Returns { visible: true, workspace, options } when the dialog is genuinely - * on screen, { visible: false } otherwise. Called on every changed screen - * frame, so the cheap whole-string reject runs first. + * Returns { visible: true, … } when either upstream layout is genuinely on + * screen, { visible: false } otherwise. Called on every changed screen frame, + * so each layout's cheap whole-string reject runs first. */ export function detectCodexTrustDialog(screen: string): CodexTrustDialogState { if (!screen) return { visible: false } - if (!QUESTION_RE.test(screen)) return { visible: false } + return detectFolderAccessLayout(screen) ?? detectYouAreInLayout(screen) ?? { visible: false } +} +function detectYouAreInLayout(screen: string): CodexTrustDialogState | null { + if (!QUESTION_RE.test(screen)) return null const lines = screen.split('\n') - let anchorIdx = -1 + + // WHY bottom-up, like the 0.156+ layout (review c of #67). The first + // structural version only required the anchor with the two option rows + // somewhere below it, so a transcript quoting the dialog verbatim (or even + // with unrelated rows between its lines) was still a live, answerable + // phantom. rust-v0.149.1's trust_directory.rs paints "Press enter to + // continue" (or the Windows "… and create a sandbox..." variant) as the + // dialog's LAST row, below an optional error paragraph, and the options are + // adjacent picker rows; a quoted copy always has the live composer below it. + let last = lines.length - 1 + while (last >= 0 && lines[last].trim() === '') last-- + if (last < 0) return null + let hintStart = last + if (!LEGACY_HINT_RE.test(lines[last])) { + if (last === 0 || lines[last - 1].trim() === '' || + !LEGACY_HINT_RE.test(`${lines[last - 1].trim()} ${lines[last].trim()}`)) return null + hintStart = last - 1 + } + + let yesIdx = -1 + for (let i = hintStart - 2; i >= 0; i--) { + if (YES_ROW_RE.test(lines[i]) && NO_ROW_RE.test(lines[i + 1])) { + yesIdx = i + break + } + } + if (yesIdx === -1) return null + let workspace: string | undefined - for (let i = 0; i < lines.length; i++) { + let anchorIdx = -1 + for (let i = yesIdx - 1; i >= 0; i--) { const m = lines[i].match(YOU_ARE_IN_RE) if (m) { anchorIdx = i @@ -91,30 +147,194 @@ export function detectCodexTrustDialog(screen: string): CodexTrustDialogState { break } } - if (anchorIdx === -1) return { visible: false } + if (anchorIdx === -1) return null - // Both option rows must appear BELOW the anchor, in order. Scanning the - // whole screen would re-admit a transcript that happens to quote them. - let yesIdx = -1 - let noIdx = -1 - for (let i = anchorIdx + 1; i < lines.length; i++) { - if (yesIdx === -1 && YES_ROW_RE.test(lines[i])) { - yesIdx = i - continue + return { + visible: true, + workspace, + options: [ + { key: '1', label: 'Yes, continue' }, + { key: '2', label: 'No, quit' }, + ], + layout: 'you-are-in', + acceptKeys: CODEX_TRUST_DIALOG_ACCEPT_KEYS, + declineKeys: CODEX_TRUST_DIALOG_DECLINE_KEYS, + } +} + +// --- 0.156+ "Folder access" layout (#65) --- +// +// Recorded from codex-cli 0.157.1 (testing/fixtures/trust-dialog-0157) and +// read against upstream `codex-rs/tui/src/onboarding/trust_directory.rs` at +// tag rust-v0.157.1: +// +// Folder access +// /path/to/folder (hard-wrapped, or …-truncated) +// +// Note: You’re in a subdirectory of a Git project. Trusting will apply +// to the repository root: (only in a Git subdirectory) +// /path/to/repo +// +// Trust this folder? Codex can read, edit, and run files here, … +// +// › 1. Trust and continue +// 2. Back to Agent Command Center +// +// (only after a failed trust write) +// +// enter continue · esc back +// +// The old anchors are all gone: no `> You are in`, no "Do you trust the +// contents", no `Yes, continue` / `No, quit`. So the previous parser returned +// not-visible for a live, blocking dialog. +// +// The same STRUCTURAL rule as the legacy layout, for the same reason (text +// that quotes the dialog must never raise a blocking modal), plus position: +// the key hint must be the LAST painted row (see detectFolderAccessLayout), +// with the nearest adjacent `1.` / `2.` pair above it and the nearest +// `Folder access` title above that. Each piece must be a whole line, and the +// option labels must be ones upstream can paint. The question paragraph is NOT an +// anchor: it has three different texts (trust / restricted / existing task) +// and it wraps at every width. +// +// WHY the labels are read from the screen and not fixed. Upstream varies both: +// option 1: "Trust and continue", or "Open restricted" for a folder saved +// as untrusted, or "Open existing task" when resuming one; +// option 2: "Quit", or "Back to Agent Command Center" when Codex is +// connected to its background server (TrustCancelAction). +// Both option-2 labels were seen locally on the same 0.157.1 binary in +// different folders. A fixed "Quit" button would lie in the second case: the +// key goes back to the overview, and Codex keeps running. +// +// The hint must agree with option 2 (`esc quit` with Quit, `esc back` with +// Back…), because upstream derives both from the same TrustCancelAction. A +// frame where they disagree is not a real render. +// +// Width floor: every anchor line is at most 33 characters +// (" 2. Back to Agent Command Center"), and upstream's own 40-column snapshots +// keep each on one row, so detection holds at 40 columns. The Windows sandbox +// hint (46 characters) wraps below 46 columns; one wrap is accepted. When +// anything wraps further, detection fails closed (not visible), the same +// failure direction as the legacy layout's floor. +const FOLDER_ACCESS_RE = /^\s*Folder access\s*$/ +const FIRST_OPTION_RE = /^\s*[›>]?\s*1\.\s*(Trust and continue|Open restricted|Open existing task)\s*$/ +const SECOND_OPTION_RE = /^\s*[›>]?\s*2\.\s*(Quit|Back to Agent Command Center)\s*$/ +// "and create sandbox" is the Windows variant of the confirm hint. +const HINT_RE = /^\s*enter continue(?: and create sandbox)?\s*·\s*esc (quit|back)\s*$/ +// Only the note's first words: the sentence wraps as early as "…of a" at 40 +// columns (upstream's long_repository_root_40x17 snapshot). +const GIT_NOTE_RE = /^\s*Note: You[’']re in a subdirectory/ +const GIT_NOTE_END_RE = /repository root:\s*$/ +// The three fixed paragraph openers. They end a path block when the dialog is +// so short that upstream drops the blank spacer rows (the 40x13 snapshot). +const PARAGRAPH_OPENER_RE = /^\s*(Trust this folder\?|Config, hooks, and exec policies|This existing task may retain)/ + +function detectFolderAccessLayout(screen: string): CodexTrustDialogState | null { + if (!screen.includes('Folder access')) return null + const lines = screen.split('\n') + + // WHY the match is anchored at the BOTTOM of the screen and read upward + // (review of #67, both reviewers). The first cut accepted the first + // `Folder access` line anywhere, then any later option pair and hint. A + // transcript that quotes this dialog verbatim (a pasted upstream .snap, a + // copied terminal frame, this repo's own plan) satisfied all of that and + // raised a blocking, ANSWERABLE phantom whose keys would then be written + // into whatever screen was really up. What a copy cannot fake is position: + // this dialog is Codex's onboarding screen, painted before any chat widget + // exists, so its key hint is the last painted row. A quoted frame inside a + // transcript always has the live composer (and footer) below it. + let last = lines.length - 1 + while (last >= 0 && lines[last].trim() === '') last-- + if (last < 0) return null + + // The hint is a wrapping paragraph. The Windows variant ("enter continue + // and create sandbox · esc quit", 46 columns with its inset) wraps onto a + // second row below 46 columns, so the last row alone or the last two rows + // joined must read as the hint. + let hintStart = last + let hintMatch = lines[last].match(HINT_RE) + if (!hintMatch && last > 0 && lines[last - 1].trim() !== '') { + hintMatch = `${lines[last - 1].trim()} ${lines[last].trim()}`.match(HINT_RE) + hintStart = last - 1 + } + if (!hintMatch) return null + const hintVerb = hintMatch[1] + + // The option pair is adjacent in every upstream render (two picker rows + // pushed back to back, no spacer) and is the nearest pair above the hint; + // only a spacer and an optional error paragraph sit between them. + let firstIdx = -1 + for (let i = hintStart - 2; i >= 0; i--) { + if (FIRST_OPTION_RE.test(lines[i]) && SECOND_OPTION_RE.test(lines[i + 1])) { + firstIdx = i + break } - if (yesIdx !== -1 && NO_ROW_RE.test(lines[i])) { - noIdx = i + } + if (firstIdx === -1) return null + const firstLabel = lines[firstIdx].match(FIRST_OPTION_RE)![1] + const secondLabel = lines[firstIdx + 1].match(SECOND_OPTION_RE)![1] + if ((hintVerb === 'quit') !== (secondLabel === 'Quit')) return null + + // The nearest `Folder access` above the options is the dialog's own title. + let anchorIdx = -1 + for (let i = firstIdx - 1; i >= 0; i--) { + if (FOLDER_ACCESS_RE.test(lines[i])) { + anchorIdx = i break } } - if (yesIdx === -1 || noIdx === -1) return { visible: false } + if (anchorIdx === -1) return null + + const { workspace, trustTarget } = readFolderAccessPaths(lines.slice(anchorIdx + 1, firstIdx)) + + return { + visible: true, + workspace: workspace ?? trustTarget, + ...(trustTarget !== undefined ? { trustTarget } : {}), + options: [ + { key: '1', label: firstLabel }, + { key: '2', label: secondLabel }, + ], + layout: 'folder-access', + acceptKeys: CODEX_TRUST_DIALOG_FOLDER_ACCESS_ACCEPT_KEYS, + declineKeys: CODEX_TRUST_DIALOG_FOLDER_ACCESS_DECLINE_KEYS, + } +} + +// Reads the folder path under `Folder access` and, in a Git subdirectory, the +// repository root under the note. +// +// WHY rows are concatenated with no separator: upstream renders a path as a +// ratatui Paragraph with `trim: false` in a (width - 4) column, so a long path +// hard-wraps at an arbitrary CHARACTER ("…/long-nested-folde" + "r" in the +// 40x24 snapshot), not at a separator. Stripping the 2-column inset and the +// right padding, then joining, restores it. The one thing this cannot restore +// is a space that sat exactly at a wrap boundary; a folder name with a space +// in precisely that column loses it. A path too tall for its rows is instead +// centre-truncated to one row with `…`, which is reported as painted. +function readFolderAccessPaths(block: string[]): { workspace?: string; trustTarget?: string } { + const joinRows = (rows: string[]) => { + const text = rows.map(row => row.replace(/^ {2}/, '').replace(/\s+$/, '')).join('') + return text.length > 0 ? text : undefined + } + const endsPathBlock = (line: string) => + line.trim() === '' || GIT_NOTE_RE.test(line) || PARAGRAPH_OPENER_RE.test(line) - const options = [ - { key: '1', label: 'Yes, continue' }, - { key: '2', label: 'No, quit' }, - ] + let i = 0 + const cwdRows: string[] = [] + while (i < block.length && !endsPathBlock(block[i])) cwdRows.push(block[i++]) + while (i < block.length && block[i].trim() === '') i++ - return { visible: true, workspace, options } + let trustTarget: string | undefined + if (i < block.length && GIT_NOTE_RE.test(block[i])) { + // The note paragraph wraps too; it ends on the row ending "root:". + while (i < block.length && !GIT_NOTE_END_RE.test(block[i])) i++ + i++ + const rootRows: string[] = [] + while (i < block.length && !endsPathBlock(block[i])) rootRows.push(block[i++]) + trustTarget = joinRows(rootRows) + } + return { workspace: joinRows(cwdRows), trustTarget } } /** @@ -138,3 +358,32 @@ export const CODEX_TRUST_DIALOG_ACCEPT_KEYS = '1' * next. */ export const CODEX_TRUST_DIALOG_DECLINE_KEYS = '2' + +/** + * The keystrokes that choose option 1 on the 0.156+ `Folder access` layout. + * + * `1` then Enter, NOT `1` alone. Upstream changed what the digit does + * (trust_directory.rs at rust-v0.157.1): `1`/`y` (SELECT_FIRST) now only MOVES + * THE HIGHLIGHT to option 1, "trust always requires an explicit Enter + * confirmation" (a terminal colour-query reply can begin with `1`), and only + * Enter (CONFIRM) acts on the highlighted row. So the legacy `'1'` would + * leave the dialog up, and a bare `'\r'` would confirm whatever happened to + * be highlighted, which may be option 2. `1` first pins the highlight, so the + * Enter that follows can only confirm option 1. + * + * Source-verified, not live-verified: answering the dialog live writes the + * trust decision into the user's own ~/.codex/config.toml, so the recording + * (testing/fixtures/trust-dialog-0157) deliberately never pressed a key. + * The upstream unit test `fragmented_terminal_response_cannot_grant_directory_trust` + * pins exactly this: digits move the highlight, Enter grants. + */ +export const CODEX_TRUST_DIALOG_FOLDER_ACCESS_ACCEPT_KEYS = '1\r' + +/** + * The keystroke that chooses option 2 on the 0.156+ layout. + * + * Still `2` alone: `2`/`n` (SELECT_SECOND) acts immediately (`handle_quit`), + * as before. What option 2 MEANS varies (quit, or back to the overview), which + * is why the condition labels its action from the screen. + */ +export const CODEX_TRUST_DIALOG_FOLDER_ACCESS_DECLINE_KEYS = '2' diff --git a/src/proxy/latestRequestBody.commitFence.test.ts b/src/proxy/latestRequestBody.commitFence.test.ts new file mode 100644 index 0000000..d914bd7 --- /dev/null +++ b/src/proxy/latestRequestBody.commitFence.test.ts @@ -0,0 +1,91 @@ +import { existsSync, mkdtempSync, readFileSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' + +import { afterEach, expect, it, vi } from 'vitest' + +// steering q96 / #70: the "never an older prompt" invariant must hold at the +// COMMIT boundary. The first fix checked the generation and THEN awaited an +// async rename. `record(B)` could run while that rename was in flight: it +// unlinked the public file, then A's rename landed and restored A as +// "latest" while B was still being written. A bundle read or a crash in that +// window labelled an older prompt current. +// +// This holds every async `rename` from fs/promises at a gate and spies the +// synchronous one, so the test can record B exactly while A's commit is +// pending, whichever API the implementation commits with. +const gate = vi.hoisted(() => ({ + held: [] as Array<() => void>, + commits: 0, +})) + +vi.mock('fs/promises', async importOriginal => { + const actual = await importOriginal() + return { + ...actual, + rename: async (from: string, to: string) => { + gate.commits += 1 + await new Promise(resolve => gate.held.push(resolve)) + return actual.rename(from, to) + }, + } +}) + +vi.mock('fs', async importOriginal => { + const actual = await importOriginal() + return { + ...actual, + renameSync: (from: string, to: string) => { + gate.commits += 1 + return actual.renameSync(from, to) + }, + } +}) + +const { LatestRequestBodySidecar } = await import('./latestRequestBody.js') + +const dirs: string[] = [] +afterEach(() => { + for (const release of gate.held.splice(0)) release() + gate.commits = 0 + for (const dir of dirs.splice(0)) rmSync(dir, { recursive: true, force: true }) +}) + +async function until(predicate: () => boolean): Promise { + const deadline = Date.now() + 5_000 + while (!predicate()) { + if (Date.now() > deadline) throw new Error('timed out') + await new Promise(resolve => setTimeout(resolve, 1)) + } +} + +it('never restores an older body when a newer one is recorded during its commit', async () => { + const dir = mkdtempSync(join(tmpdir(), 'cxh-latest-body-fence-')) + dirs.push(dir) + const sidecar = new LatestRequestBodySidecar(join(dir, 'proxy-events.jsonl')) + + sidecar.record('req-a', 'responses', Buffer.from('OLDER prompt A')) + // A has written its temp file and reached its commit. + await until(() => gate.commits === 1) + + // B arrives now. It is large, so its own write spans many turns and cannot + // commit within the window below. + sidecar.record('req-b', 'responses', Buffer.alloc(4 * 1024 * 1024, 0x62)) + for (const release of gate.held.splice(0)) release() + // Let A's commit, if it was still pending, land. + await new Promise(resolve => setTimeout(resolve, 20)) + + // Before B commits, the public file is absent or B: never A. + if (existsSync(sidecar.path)) { + expect(readFileSync(sidecar.path, 'utf8')).not.toContain(Buffer.from('OLDER prompt A').toString('base64')) + } + + // Let every remaining write finish, releasing any rename that is held. + const releaser = setInterval(() => { for (const release of gate.held.splice(0)) release() }, 1) + try { + await sidecar.flush() + } finally { + clearInterval(releaser) + } + expect(readFileSync(sidecar.path, 'utf8')).toContain('"requestId":"req-b"') +}) diff --git a/src/proxy/latestRequestBody.test.ts b/src/proxy/latestRequestBody.test.ts new file mode 100644 index 0000000..4c9117f --- /dev/null +++ b/src/proxy/latestRequestBody.test.ts @@ -0,0 +1,84 @@ +import { existsSync, mkdtempSync, readFileSync, readdirSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' + +import { afterEach, expect, it } from 'vitest' + +import { LatestRequestBodySidecar } from './latestRequestBody.js' + +// #70 review a: "never an older prompt" must hold at every instant a bundle +// can read the file, not only once writes settle, and a large body must not +// be encoded in one main-process turn. + +const dirs: string[] = [] +afterEach(() => { for (const dir of dirs.splice(0)) rmSync(dir, { recursive: true, force: true }) }) +function sidecarIn(): LatestRequestBodySidecar { + const dir = mkdtempSync(join(tmpdir(), 'cxh-latest-body-unit-')) + dirs.push(dir) + return new LatestRequestBodySidecar(join(dir, 'proxy-events.jsonl')) +} +const bodyOf = (path: string) => Buffer.from((JSON.parse(readFileSync(path, 'utf8')) as { body_b64: string }).body_b64, 'base64') + +it('removes the older body the moment a newer one is recorded', async () => { + const sidecar = sidecarIn() + sidecar.record('req-1', 'responses', Buffer.from('older prompt')) + await sidecar.flush() + expect(existsSync(sidecar.path)).toBe(true) + + sidecar.record('req-2', 'responses', Buffer.from('newer prompt')) + // Synchronously, before any write: a reader (or a crash) now sees no body, + // never the older one. + expect(existsSync(sidecar.path)).toBe(false) + await sidecar.flush() + expect(bodyOf(sidecar.path).toString()).toBe('newer prompt') +}) + +it('never lets a superseded in-flight write rename over a newer body', async () => { + const sidecar = sidecarIn() + // A large body takes several turns to encode, so the next record lands + // while it is still in flight. + const large = Buffer.alloc(4 * 1024 * 1024, 0x61) + sidecar.record('req-1', 'responses', large) + sidecar.record('req-2', 'responses', Buffer.from('newest prompt')) + await sidecar.flush() + + expect(bodyOf(sidecar.path).toString()).toBe('newest prompt') + // The superseded write left no temp file behind. + expect(readdirSync(join(sidecar.path, '..')).filter(name => name.endsWith('.tmp'))).toEqual([]) +}) + +it('writes a multi-slice body that round-trips byte for byte', async () => { + const sidecar = sidecarIn() + // Not a multiple of the 768 KiB slice, and arbitrary bytes (a zstd frame is + // binary), so slice boundaries and padding are both exercised. + const body = Buffer.from(Array.from({ length: 2 * 1024 * 1024 + 7 }, (_, i) => (i * 131 + 7) % 256)) + sidecar.record('req-1', 'responses', body) + await sidecar.flush() + + expect(bodyOf(sidecar.path).equals(body)).toBe(true) + expect(readFileSync(sidecar.path, 'utf8').split('\n').filter(Boolean)).toHaveLength(1) +}) + +it('does not resurrect an in-flight older body after a newer one was too big to keep', async () => { + // The newer body is over the cap, so nothing replaces the file; only the + // superseded write's own generation check stops it renaming the older body + // back in as "latest". + const sidecar = sidecarIn() + sidecar.record('req-1', 'responses', Buffer.alloc(4 * 1024 * 1024, 0x61)) + sidecar.record('req-2', 'responses', Buffer.alloc(16 * 1024 * 1024 + 1, 0x62)) + await sidecar.flush() + + expect(existsSync(sidecar.path)).toBe(false) +}) + +it('sweeps a crash-left temp file on the next successful commit', async () => { + // #70 review c: a crash between writing and renaming leaves a temp file + // holding a full prompt. It used to be cleaned only after a failed write. + const sidecar = sidecarIn() + const { writeFileSync } = await import('node:fs') + writeFileSync(`${sidecar.path}.99999.7.tmp`, 'crash-left prompt body') + sidecar.record('req-1', 'responses', Buffer.from('prompt')) + await sidecar.flush() + expect(readdirSync(join(sidecar.path, '..')).filter(name => name.endsWith('.tmp'))).toEqual([]) + expect(bodyOf(sidecar.path).toString()).toBe('prompt') +}) diff --git a/src/proxy/latestRequestBody.ts b/src/proxy/latestRequestBody.ts new file mode 100644 index 0000000..b1de04f --- /dev/null +++ b/src/proxy/latestRequestBody.ts @@ -0,0 +1,189 @@ +import { createWriteStream, renameSync, unlinkSync } from 'fs' +import { readdir, rm } from 'fs/promises' +import { basename, dirname, join } from 'path' + +// The newest main-turn Responses request body, kept in a sidecar next to the +// proxy events file (agent-code#1336). +// +// WHY a sidecar at all: Agent Code's debug bundle carries the last 5 MiB of +// `proxy-events.jsonl` (plus the rotated `.1`). On Codex the response chunks +// are the bulk of that file, so a turn that streams more than 5 MiB, or a run +// of `/v1/models` refreshes, pushes the request event (and its `body_b64`) +// out of the tail. The bundle then holds recent chunks and no prompt, which +// is the one thing a bug report needs. Measured on the live corpus (review C +// of agent-code#1332): the last inline body lay more than 5 MiB before EOF in +// 40 of 65 Codex files (10 of 65 once chunks are base64). Rotation cannot +// help, because the problem is the tail, not the file size. +// +// WHY this name and line shape: it is the Claude addon's sidecar +// (claude-code-headless `mitmAddon.py`, agent-code#1273), and Agent Code's +// `readLatestRequestBody` already appends `/latest-request-body.json` +// to the bundle's proxy section for ANY provider. One JSON line with its own +// `kind`, so nothing that replays the events file mistakes it for a request. +// The body is the raw on-wire bytes, exactly as the `request` event's +// `body_b64` (zstd-compressed on current Codex; readers detect the frame). +// +// The INVARIANT differs from Claude's, on purpose. Claude's sidecar holds the +// newest body that is NOT in the log (its log omits bodies past a budget). +// Codex never omits bodies up to the 2 MiB inline cap, and the loss here is +// the bundle's tail, which the proxy cannot see. So this sidecar holds the +// newest main-turn body, even when the log also has it (at most one body +// appears twice in a bundle). +// +// WHICH requests (the caller decides, see ResponsesProxy): `responses*` +// endpoints with a body, excluding temporary structured turns. Codex 0.157's +// title generation (tui `thread_title.rs`) runs an ephemeral thread whose +// request carries an output schema (`text.format`, name +// `codex_output_schema`); ordinary TUI turns never do. Letting it in replaced +// the main prompt with a 960-byte title prompt (#70 review a). Subagent calls +// (`x-openai-subagent`, except `compact`) are skipped for the same reason +// (#70 review b). A body whose +// shape cannot be read (not JSON, or a zstd frame over the decode bound) is +// still recorded: an unclassified prompt is better evidence than none. +// +// "NEVER AN OLDER PROMPT" (#70 review a): three rules make it hold at every +// instant a reader can look, not only after writes settle. +// 1. A newer body removes the real-name file SYNCHRONOUSLY when it is +// recorded. `unlinkSync` is one metadata call, not a data write. From then +// on a reader sees either nothing or the newer body, including after a +// crash anywhere in the write. +// 2. A write that has been superseded by a newer record never renames: it +// checks its generation and publishes with `renameSync` in the same +// synchronous turn, so no record() can slip between check and commit +// (steering q96), and a superseded write discards its temp file. +// 3. Only the newest pending body is kept. Older pending bodies are dropped +// unwritten, so memory holds at most one body in flight and one pending. +// If the directory refuses the unlink (permissions), nothing here can make the +// file current; that failure is not reported, as for every mirror failure. +// +// MAIN-PROCESS COST (#70 review a): the proxy runs on Agent Code's main +// process. The only synchronous calls are two metadata operations (the +// unlink in record, the rename at commit); every data byte is written +// asynchronously. A 16 MiB body is 21.3 MiB of base64, measured at up to 376 ms when +// encoded in one call. It is encoded in slices of ENCODE_SLICE_BYTES, each in +// its own turn after the stream drains, so no single turn does more than a +// slice. + +export const LATEST_REQUEST_BODY_FILE_NAME = 'latest-request-body.json' + +// Same cap as the Claude addon's sidecar (`_LATEST_BODY_CAP`). Larger than the +// 2 MiB inline cap on purpose: a body too big to inline is exactly the one the +// events file cannot show at all. Such bodies are persisted ONLY here +// (SECURITY.md says so). +export const LATEST_REQUEST_BODY_CAP = 16 * 1024 * 1024 + +// A multiple of 3, so every slice but the last encodes to base64 without +// padding and the slices concatenate into one valid base64 string. +const ENCODE_SLICE_BYTES = 3 * 256 * 1024 + +type Pending = { generation: number; requestId: string; endpoint: string; body: Buffer } + +export class LatestRequestBodySidecar { + readonly path: string + private generation = 0 + private pending: Pending | null = null + private inFlight: Promise | null = null + private idleWaiters: Array<() => void> = [] + + constructor(eventsFile: string) { + this.path = join(dirname(eventsFile), LATEST_REQUEST_BODY_FILE_NAME) + } + + record(requestId: string, endpoint: string, body: Buffer): void { + this.generation += 1 + // Rule 1: from now on the old body is not "latest", on disk too. + try { unlinkSync(this.path) } catch { /* absent, or the directory refuses */ } + this.pending = body.length > LATEST_REQUEST_BODY_CAP + ? null + : { generation: this.generation, requestId, endpoint, body } + this.kick() + } + + /** Resolves once every recorded body has been written, or dropped. */ + flush(): Promise { + if (!this.inFlight && !this.pending) return Promise.resolve() + return new Promise(resolve => this.idleWaiters.push(resolve)) + } + + private kick(): void { + if (this.inFlight) return + const next = this.pending + this.pending = null + if (!next) { + for (const resolve of this.idleWaiters.splice(0)) resolve() + return + } + this.inFlight = this.write(next).finally(() => { + this.inFlight = null + this.kick() + }) + } + + private async write(entry: Pending): Promise { + // A per-write temp name, so a crash mid-write never leaves a half body + // under the real name. + const temp = `${this.path}.${process.pid}.${entry.generation}.tmp` + try { + await writeEncoded(temp, entry) + // Rule 2, enforced AT the commit (steering q96). The generation check + // and the publish run in ONE synchronous turn, so no `record()` can run + // between them. The first version checked, then awaited an async + // `rename`: a `record(B)` in that gap unlinked the public file, and A's + // rename then landed and restored A as "latest" while B was still being + // written. `renameSync` is one metadata call (like the `unlinkSync` in + // record), not a data write; the body itself is still written + // asynchronously in slices above, so a large body never blocks a turn. + if (entry.generation !== this.generation) { + await rm(temp, { force: true }) + return + } + renameSync(temp, this.path) + // Temp files a crash left between write and rename hold full prompt + // bodies; sweep them on a successful commit too, not only after a + // failure (#70 review c). Not awaited in the commit's turn. + await this.removeTempFiles() + } catch { + await rm(temp, { force: true }).catch(() => {}) + await this.removeTempFiles() + } + } + + private async removeTempFiles(): Promise { + // Temp files a crash between write and rename left behind. + try { + const prefix = `${basename(this.path)}.` + for (const name of await readdir(dirname(this.path))) { + if (name.startsWith(prefix) && name.endsWith('.tmp')) await rm(join(dirname(this.path), name), { force: true }).catch(() => {}) + } + } catch { /* directory gone: nothing to clean */ } + } +} + +async function writeEncoded(path: string, entry: Pending): Promise { + const stream = createWriteStream(path, { encoding: 'utf-8' }) + const done = new Promise((resolve, reject) => { + stream.once('error', reject) + stream.once('finish', resolve) + }) + // Observed here so an error while a slice is still being encoded is not an + // unhandled rejection; it is re-raised by the race below and the final await. + done.catch(() => {}) + const put = async (text: string): Promise => { + // Raced with `done`: a stream that errors never emits 'drain'. + if (!stream.write(text)) await Promise.race([new Promise(resolve => stream.once('drain', resolve)), done]) + // Yield so the next slice's encoding runs in its own turn. + await new Promise(resolve => setImmediate(resolve)) + } + try { + const head = JSON.stringify({ kind: 'request-body-latest', requestId: entry.requestId, endpoint: entry.endpoint }) + await put(`${head.slice(0, -1)},"body_b64":"`) + for (let offset = 0; offset < entry.body.length; offset += ENCODE_SLICE_BYTES) { + await put(entry.body.subarray(offset, offset + ENCODE_SLICE_BYTES).toString('base64')) + } + stream.end('"}\n') + } catch (error) { + stream.destroy() + throw error + } + await done +} diff --git a/src/proxy/responsesProxy.latestBody.test.ts b/src/proxy/responsesProxy.latestBody.test.ts new file mode 100644 index 0000000..61ef2f4 --- /dev/null +++ b/src/proxy/responsesProxy.latestBody.test.ts @@ -0,0 +1,187 @@ +import { existsSync, mkdtempSync, readFileSync, rmSync } from 'node:fs' +import { createServer, request as httpRequest, type Server } from 'node:http' +import { tmpdir } from 'node:os' +import { join } from 'node:path' + +import { afterEach, expect, it } from 'vitest' + +import { ResponsesProxy } from './responsesProxy.js' + +// agent-code#1336: Agent Code's debug bundle tails the newest 5 MiB of +// proxy-events.jsonl. On Codex the response chunks are the bulk, so after a +// long stream the request event and its body_b64 fall out of the tail and the +// bundle has no prompt (40 of 65 recorded files, review C of #1332). The +// Claude addon solved the same loss with `latest-request-body.json`, which +// Agent Code already appends for any provider. These drive REAL requests +// through the proxy to a local upstream and read what lands on disk. + +// Literal on purpose, not imported: Agent Code's reader looks for this exact +// name, and the test must fail on a proxy that never writes it. +const LATEST_REQUEST_BODY_FILE_NAME = 'latest-request-body.json' +// The Claude addon's sidecar cap, mirrored. +const LATEST_REQUEST_BODY_CAP = 16 * 1024 * 1024 + +const cleanup: Array<() => Promise | void> = [] +afterEach(async () => { for (const step of cleanup.splice(0).reverse()) await step() }) + +async function startProxy(): Promise<{ proxy: ResponsesProxy; eventsFile: string; sidecar: string }> { + const upstream: Server = createServer((req, res) => { + req.resume() + req.on('end', () => { + res.writeHead(200, { 'content-type': 'text/event-stream' }) + res.end('event: response.completed\ndata: {"type":"response.completed"}\n\n') + }) + }) + await new Promise(resolve => upstream.listen(0, '127.0.0.1', resolve)) + cleanup.push(() => new Promise(resolve => { upstream.closeAllConnections(); upstream.close(() => resolve()) })) + const address = upstream.address() + if (!address || typeof address === 'string') throw new Error('upstream did not bind') + const dir = mkdtempSync(join(tmpdir(), 'cxh-latest-body-')) + cleanup.push(() => rmSync(dir, { recursive: true, force: true })) + const eventsFile = join(dir, 'proxy-events.jsonl') + const proxy = await ResponsesProxy.create({ + upstreamBaseUrl: `http://127.0.0.1:${address.port}/v1`, + authMode: 'apikey', + eventsFile, + }) + cleanup.push(() => proxy.stop()) + return { proxy, eventsFile, sidecar: join(dir, LATEST_REQUEST_BODY_FILE_NAME) } +} + +function send(proxy: ResponsesProxy, path: string, method: string, body?: Buffer): Promise { + return new Promise((resolve, reject) => { + const client = httpRequest(`${proxy.info.proxyBaseUrl}${path}`, { method, headers: { 'content-type': 'application/json' } }, res => { + res.resume() + res.on('end', () => resolve()) + }) + client.on('error', reject) + client.end(body) + }) +} + +const sidecarLine = (path: string) => JSON.parse(readFileSync(path, 'utf8').trim()) as Record + +it('keeps the newest Responses request body next to the events file', async () => { + const { proxy, sidecar } = await startProxy() + const first = Buffer.from(JSON.stringify({ model: 'gpt-5', input: [{ role: 'user', content: 'first prompt' }] })) + const second = Buffer.from(JSON.stringify({ model: 'gpt-5', input: [{ role: 'user', content: 'second prompt' }] })) + + await send(proxy, '/responses', 'POST', first) + await send(proxy, '/responses', 'POST', second) + await proxy.flushMirror() + + const line = sidecarLine(sidecar) + expect(line).toMatchObject({ kind: 'request-body-latest', requestId: 'req-2', endpoint: 'responses' }) + expect(Buffer.from(line.body_b64 as string, 'base64').equals(second)).toBe(true) + // One line: the reader appends it verbatim to the events tail. + expect(readFileSync(sidecar, 'utf8').split('\n').filter(Boolean)).toHaveLength(1) +}) + +it('is not replaced by a bodiless /models refresh', async () => { + const { proxy, sidecar } = await startProxy() + const prompt = Buffer.from(JSON.stringify({ input: [{ role: 'user', content: 'the prompt' }] })) + await send(proxy, '/responses', 'POST', prompt) + await send(proxy, '/models', 'GET') + await proxy.flushMirror() + + expect(Buffer.from(sidecarLine(sidecar).body_b64 as string, 'base64').equals(prompt)).toBe(true) +}) + +it('removes the sidecar rather than keep an older prompt when the newest body is over the cap', async () => { + const { proxy, sidecar } = await startProxy() + await send(proxy, '/responses', 'POST', Buffer.from('{"input":"older prompt"}')) + await proxy.flushMirror() + expect(existsSync(sidecar)).toBe(true) + + await send(proxy, '/responses', 'POST', Buffer.alloc(LATEST_REQUEST_BODY_CAP + 1, 0x20)) + await proxy.flushMirror() + expect(existsSync(sidecar)).toBe(false) +}) + +// #70 review a. Codex 0.157 title generation runs an ephemeral thread whose +// request carries an output schema (tui `thread_title.rs` through +// codex-api `create_text_param_for_request`: text.format, name +// `codex_output_schema`). It must not replace the main prompt. +it('keeps the main prompt when a title-generation request follows it', async () => { + const { proxy, sidecar } = await startProxy() + const prompt = Buffer.from(JSON.stringify({ input: [{ role: 'user', content: 'the main prompt' }], tools: [{ type: 'function' }] })) + const title = Buffer.from(JSON.stringify({ + input: [{ role: 'user', content: 'Generate a concise, single-line task title' }], + text: { format: { type: 'json_schema', strict: true, name: 'codex_output_schema', schema: {} } }, + })) + await send(proxy, '/responses', 'POST', prompt) + await send(proxy, '/responses', 'POST', title) + await proxy.flushMirror() + + expect(Buffer.from(sidecarLine(sidecar).body_b64 as string, 'base64').equals(prompt)).toBe(true) +}) + +it('records a compaction request, which also carries the conversation', async () => { + const { proxy, sidecar } = await startProxy() + const compact = Buffer.from(JSON.stringify({ input: [{ role: 'user', content: 'compact me' }] })) + await send(proxy, '/responses/compact', 'POST', compact) + await proxy.flushMirror() + + expect(sidecarLine(sidecar)).toMatchObject({ endpoint: 'responses/compact' }) +}) + +// #70 review b: Codex tags every non-main Responses call with +// x-openai-subagent (codex-api requests/headers.rs). A spawned worker's task +// must not replace the top-level prompt; a compaction must still be kept. +function sendWithSubagent(proxy: ResponsesProxy, path: string, body: Buffer, subagent: string): Promise { + return new Promise((resolve, reject) => { + const client = httpRequest(`${proxy.info.proxyBaseUrl}${path}`, { + method: 'POST', + headers: { 'content-type': 'application/json', 'x-openai-subagent': subagent }, + }, res => { + res.resume() + res.on('end', () => resolve()) + }) + client.on('error', reject) + client.end(body) + }) +} + +it('keeps the main prompt when a spawned subagent sends its own request', async () => { + const { proxy, sidecar } = await startProxy() + const prompt = Buffer.from(JSON.stringify({ input: [{ role: 'user', content: 'the main prompt' }] })) + await send(proxy, '/responses', 'POST', prompt) + await sendWithSubagent(proxy, '/responses', Buffer.from('{"input":"worker task"}'), 'thread_spawn') + await sendWithSubagent(proxy, '/responses', Buffer.from('{"input":"review task"}'), 'review') + await proxy.flushMirror() + + expect(Buffer.from(sidecarLine(sidecar).body_b64 as string, 'base64').equals(prompt)).toBe(true) +}) + +it('still records a compaction tagged as the compact subagent', async () => { + const { proxy, sidecar } = await startProxy() + await send(proxy, '/responses', 'POST', Buffer.from('{"input":"the main prompt"}')) + await sendWithSubagent(proxy, '/responses', Buffer.from('{"input":"compact me"}'), 'compact') + await proxy.flushMirror() + + expect(Buffer.from(sidecarLine(sidecar).body_b64 as string, 'base64').toString()).toBe('{"input":"compact me"}') +}) + +it('keeps a body of exactly the cap', async () => { + const { proxy, sidecar } = await startProxy() + const body = Buffer.alloc(LATEST_REQUEST_BODY_CAP, 0x20) + await send(proxy, '/responses', 'POST', body) + await proxy.flushMirror() + + expect(Buffer.from(sidecarLine(sidecar).body_b64 as string, 'base64').length).toBe(LATEST_REQUEST_BODY_CAP) +}) + +// #70 review c: the endpoint filter is the only guard against auxiliary +// POSTs with a body (memory summarization, the web-search sideband) replacing +// the prompt. Nothing pinned it: `/models` is bodiless, so it is skipped +// before the endpoint is ever consulted. +it('keeps the main prompt when a bodied non-Responses request follows it', async () => { + const { proxy, sidecar } = await startProxy() + const prompt = Buffer.from(JSON.stringify({ input: [{ role: 'user', content: 'the main prompt' }] })) + await send(proxy, '/responses', 'POST', prompt) + await send(proxy, '/memories/trace_summarize', 'POST', Buffer.from('{"traces":["memory summary input"]}')) + await send(proxy, '/alpha/search', 'POST', Buffer.from('{"query":"a web search"}')) + await proxy.flushMirror() + + expect(Buffer.from(sidecarLine(sidecar).body_b64 as string, 'base64').equals(prompt)).toBe(true) +}) diff --git a/src/proxy/responsesProxy.ts b/src/proxy/responsesProxy.ts index b6355c2..f87daaf 100644 --- a/src/proxy/responsesProxy.ts +++ b/src/proxy/responsesProxy.ts @@ -7,6 +7,7 @@ import { homedir } from 'os' import { join } from 'path' import { EventsMirror } from './eventsMirror.js' +import { LatestRequestBodySidecar } from './latestRequestBody.js' import { decompressZstdBounded } from './zstd.js' // Local HTTP proxy for Codex's Responses API. @@ -214,6 +215,15 @@ export type CodexRequestShape = { input_items_count: number | null tools_count: number | null has_reasoning: boolean + /** True when the request asks for structured output (`text.format`, an + * output schema). In the TUI that Agent Code runs, Codex sets it on + * temporary structured turns such as 0.157 title generation + * (`thread_title.rs`). It is a per-turn option upstream, though (exec, + * app-server `turn_start`), so a main turn COULD carry one; Agent Code + * sends none today. The latest-request-body sidecar skips these (#70 + * reviews a, c): a structured main turn would leave the sidecar one turn + * behind, which is the accepted cost of keeping title prompts out. */ + has_output_schema: boolean client_metadata: { thread_id: string | null session_id: string | null @@ -247,6 +257,8 @@ function extractRequestShape(body: Buffer): CodexRequestShape | null { const input = Array.isArray(obj.input) ? obj.input.length : null const tools = Array.isArray(obj.tools) ? obj.tools.length : null const hasReasoning = obj.reasoning != null && typeof obj.reasoning === 'object' + const text = obj.text && typeof obj.text === 'object' ? obj.text as Record : null + const hasOutputSchema = text?.format != null && typeof text.format === 'object' const metadataObject = obj.client_metadata const metadata = metadataObject && typeof metadataObject === 'object' ? metadataObject as Record @@ -268,6 +280,7 @@ function extractRequestShape(body: Buffer): CodexRequestShape | null { input_items_count: input, tools_count: tools, has_reasoning: hasReasoning, + has_output_schema: hasOutputSchema, client_metadata: clientMetadata, provider_session_id: threadId !== null && threadId === sessionId ? threadId : null, @@ -392,6 +405,9 @@ export class ResponsesProxy extends EventEmitter { // bounded async stream; see eventsMirror.ts for why it is no longer a // synchronous append per event (agent-code#372). private readonly mirror: EventsMirror | null + // The newest Responses request body, next to the events file, so a debug + // bundle has the prompt even when the events tail does not (latestRequestBody.ts). + private readonly latestBody: LatestRequestBodySidecar | null constructor( info: CodexResponsesProxyInfo, @@ -403,6 +419,8 @@ export class ResponsesProxy extends EventEmitter { this.mirror = eventsFile ? new EventsMirror(eventsFile, { maxFileBytes: mirrorOptions.eventsFileMaxBytes }) : null + // Same opt-in as the mirror: no events file, no sidecar (agent-code#1336). + this.latestBody = eventsFile ? new LatestRequestBodySidecar(eventsFile) : null } // Override emit to mirror events into the on-disk JSONL. Done at @@ -417,9 +435,10 @@ export class ResponsesProxy extends EventEmitter { return super.emit(event, ...args) } - /** Resolves once every mirrored event so far has reached the OS. */ + /** Resolves once every mirrored event, and the latest-request-body sidecar, + * has reached the OS. */ flushMirror(): Promise { - return this.mirror?.flush() ?? Promise.resolve() + return Promise.all([this.mirror?.flush(), this.latestBody?.flush()]).then(() => undefined) } static async create(options: Options = {}): Promise { @@ -526,6 +545,7 @@ export class ResponsesProxy extends EventEmitter { await this.stopServer() } finally { await this.mirror?.close() + await this.latestBody?.flush() } } @@ -698,6 +718,24 @@ export class ResponsesProxy extends EventEmitter { ...(bodyB64 !== undefined ? { body_b64: bodyB64 } : {}), ...(requestShape !== null ? { request_shape: requestShape } : {}), }) + // The newest MAIN-turn prompt, for debug bundles whose events tail no + // longer holds this request (agent-code#1336). Temporary structured turns + // (title generation) are skipped: see latestRequestBody.ts for why, and + // for why a body whose shape could not be read is still kept (a prompt we + // cannot classify is better evidence than none). + // Subagent calls are skipped too (#70 review b): Codex tags every + // non-main Responses call with `x-openai-subagent` (codex-api + // requests/headers.rs and responses_metadata.rs: review, compact, + // collab_spawn, guardian, memory_consolidation, or a custom label; the + // rule below is label-agnostic, so the exact list does not matter). A spawned worker's task has + // no output schema, so without this it replaced the top-level prompt in + // multi-agent sessions. `compact` is kept: a compaction request carries + // the main conversation, and it is what a bundle after compaction needs. + const subagent = req.headers['x-openai-subagent'] + const auxiliary = subagent !== undefined && subagent !== 'compact' + if (body && body.length > 0 && endpoint.startsWith('responses') && requestShape?.has_output_schema !== true && !auxiliary) { + this.latestBody?.record(requestId, endpoint, body) + } const abort = new AbortController() const headersTimer = setTimeout(() => { diff --git a/src/transcript/SubmittedPromptInput.recorded.test.ts b/src/transcript/SubmittedPromptInput.recorded.test.ts index 637815a..b001971 100644 --- a/src/transcript/SubmittedPromptInput.recorded.test.ts +++ b/src/transcript/SubmittedPromptInput.recorded.test.ts @@ -1,7 +1,7 @@ import { readFileSync } from 'node:fs' import { fileURLToPath } from 'node:url' -import { describe, expect, it } from 'vitest' +import { beforeAll, describe, expect, it } from 'vitest' import type { StableTerminalFrame } from '../terminal/HeadlessTerminal.js' import { @@ -10,6 +10,7 @@ import { type SubmittedPromptInputContext, } from './SubmittedPromptInput.js' import { + type CodexPromptInputProfile, prepareCodex01491PromptInputProfile, } from './prompt-input/CodexPromptInputProfile.js' import { @@ -128,535 +129,650 @@ type RecordedConsumer = ( context: RecordedInputContext, ) => string[] -const fixturePath = fileURLToPath(new URL( - '../../testing/fixtures/prompt-input/codex-01491-recorded.json', - import.meta.url, -)) -const catalogPath = fileURLToPath(new URL( - '../../testing/fixtures/prompt-input/catalog.md', - import.meta.url, -)) -const configSourcePath = fileURLToPath(new URL( - '../../testing/fixtures/prompt-input/codex-01491-config-source.json', - import.meta.url, -)) -const appServerFixturePath = fileURLToPath(new URL( - '../../testing/fixtures/prompt-input/codex-01491-app-server-fixture.mjs', - import.meta.url, -)) -const corpus = JSON.parse( - readFileSync(fixturePath, 'utf8'), -) as RecordedPromptInputCorpus -const catalog = readFileSync(catalogPath, 'utf8') -const configSource = JSON.parse(readFileSync(configSourcePath, 'utf8')) as { - schemaVersion: number - upstreamTag: string +// WHY one suite per recorded Codex version (#63). The prompt-input profile is +// issued only for versions with their own full live recording, so every +// recording must pass the same executable contract. Only provenance differs +// per version: the recorded corpus, its config/read projection, and the +// audited upstream config source. Everything below the spec is shared, so a +// behavioural difference between versions shows up as a failing case rather +// than as a second hand-maintained copy of the rules. +type RecordedCorpusSpec = { + title: string + corpusFile: string + configSourceFile: string + configReadRecordingFile: string + provider: { cliVersion: string; binarySha256: string; upstreamTag: string } upstreamCommitSha: string - recordedCases: Record - files: Array<{ - path: string - sha256: string - coordinates: Array<{ startLine: number; endLine: number; claim: string }> - }> -} -const recordedProfilePreparation = await prepareRecordedProfile() -if (!recordedProfilePreparation.ok) { - throw new Error('recorded config/read profile fixture was refused') + configSourceFiles: Array<[string, string]> + /** The two popup-Enter cases exist only in corpora recorded on 0.156+. */ + recordsPopupEnterCases: boolean } -const recordedIssuedProfile = recordedProfilePreparation.profile - -const recordedCaseIds = [ - 'trust-action-then-submit', - 'combining-grapheme-backspace', - 'mixed-cjk-ctrl-w', - 'repeated-line-boundaries', - 'remapped-kill-line-start', - 'vim-normal-default', - 'unbound-submit-enter', - 'modal-ctrl-c-preserves-draft', - 'tab-footer-spoof-skill-popup', - 'active-footer-tab-queue', - 'narrow-soft-wrap-resize-redraw', - 'unchanged-redraw-after-edit', - 'ordinary-modal-sentinel-draft', - 'ordinary-vim-sentinel-cwd', - 'lower-layer-keymap-valid-control', - 'lower-layer-keymap-issued-profile-conflict', -] as const -const priorReplayCaseIds = new Set(recordedCaseIds.slice(0, 10)) - -const unicodeBoundaryCases = new Set([ - 'combining-grapheme-backspace', - 'mixed-cjk-ctrl-w', -]) - -const casesById = corpus.cases.map(recordedCase => [ - recordedCase.id, - recordedCase, -] as const) -const priorReplayCasesById = casesById.filter(([caseId]) => - priorReplayCaseIds.has(caseId as typeof recordedCaseIds[number]), -) - -function contextFor( - recordedCase: RecordedPromptInputCase, - screenRows?: string[], -): RecordedInputContext { - const screenBeforeWrite = screenRows?.join('\n') - return { - // WHY the old helper must remain in this replay until Stage 27 replaces its - // whole-screen guess. Feeding the exact recorded rows makes the popup-spoof - // case fail for the real reason: transcript prose currently masquerades as - // the active bottom footer, while the genuine queue footer remains green. - tabBehavior: inferCodexTabBehavior(screenBeforeWrite ?? ''), - screenBeforeWrite, - inputProfile: recordedCase.configClass === 'recorded-default-01491' - ? recordedIssuedProfile - : { - cliVersion: corpus.provider.cliVersion, - upstreamTag: corpus.provider.upstreamTag, - configClass: recordedCase.configClass, - configOverrides: [...recordedCase.configOverrides], - }, - } -} +// Recorded only on 0.156+ (review a of codex-headless#69): Enter while a popup +// that paints ABOVE the composer owns it. The 0.149.1 corpus predates them, and +// that binary can no longer be re-recorded here. +const POPUP_ENTER_CASE_IDS = [ + 'slash-popup-enter-selects-command', + 'file-popup-enter-inserts-mention', +] as const -function recordedScreenBeforeChunk( - recordedCase: RecordedPromptInputCase, - chunkIndex: number, -): string[] | undefined { - const finalChunkIndex = recordedCase.inputChunks.length - 1 - if (chunkIndex === finalChunkIndex) return recordedCase.screenBeforeFinalWrite - - // WHY these are the only retained intermediate frame boundaries in the - // sanitized corpus. The Stage 25 recorder captured history search after - // Ctrl+R, so its popup is the pre-Ctrl+C surface. It captured the active-turn - // footer before typing the queued draft. Inventing frames for every ordinary - // character would turn provider recordings back into imagined fixtures. - if (recordedCase.id === 'modal-ctrl-c-preserves-draft' && chunkIndex === 2) { - return recordedCase.popup +function defineRecordedCorpusSuite(spec: RecordedCorpusSpec): void { + const fixturePath = fileURLToPath(new URL( + `../../testing/fixtures/prompt-input/${spec.corpusFile}`, + import.meta.url, + )) + const catalogPath = fileURLToPath(new URL( + '../../testing/fixtures/prompt-input/catalog.md', + import.meta.url, + )) + const configSourcePath = fileURLToPath(new URL( + `../../testing/fixtures/prompt-input/${spec.configSourceFile}`, + import.meta.url, + )) + const appServerFixturePath = fileURLToPath(new URL( + '../../testing/fixtures/prompt-input/codex-01491-app-server-fixture.mjs', + import.meta.url, + )) + const corpus = JSON.parse( + readFileSync(fixturePath, 'utf8'), + ) as RecordedPromptInputCorpus + const catalog = readFileSync(catalogPath, 'utf8') + const configSource = JSON.parse(readFileSync(configSourcePath, 'utf8')) as { + schemaVersion: number + upstreamTag: string + upstreamCommitSha: string + recordedCases: Record + files: Array<{ + path: string + sha256: string + coordinates: Array<{ startLine: number; endLine: number; claim: string }> + }> } - if (recordedCase.id === 'active-footer-tab-queue' && chunkIndex === 0) { - return recordedCase.activeTurnFooter + // Issued in beforeAll (below) from this version's own config/read recording. + let recordedIssuedProfile!: CodexPromptInputProfile + + const recordedCaseIds = [ + 'trust-action-then-submit', + 'combining-grapheme-backspace', + 'mixed-cjk-ctrl-w', + 'repeated-line-boundaries', + 'remapped-kill-line-start', + 'vim-normal-default', + 'unbound-submit-enter', + 'modal-ctrl-c-preserves-draft', + 'tab-footer-spoof-skill-popup', + 'active-footer-tab-queue', + 'narrow-soft-wrap-resize-redraw', + 'unchanged-redraw-after-edit', + 'ordinary-modal-sentinel-draft', + 'ordinary-vim-sentinel-cwd', + 'lower-layer-keymap-valid-control', + 'lower-layer-keymap-issued-profile-conflict', + ] as const + + const priorReplayCaseIds = new Set(recordedCaseIds.slice(0, 10)) + + const unicodeBoundaryCases = new Set([ + 'combining-grapheme-backspace', + 'mixed-cjk-ctrl-w', + ]) + + const casesById = corpus.cases.map(recordedCase => [ + recordedCase.id, + recordedCase, + ] as const) + const priorReplayCasesById = casesById.filter(([caseId]) => + priorReplayCaseIds.has(caseId as typeof recordedCaseIds[number]), + ) + + function contextFor( + recordedCase: RecordedPromptInputCase, + screenRows?: string[], + ): RecordedInputContext { + const screenBeforeWrite = screenRows?.join('\n') + return { + // WHY the old helper must remain in this replay until Stage 27 replaces its + // whole-screen guess. Feeding the exact recorded rows makes the popup-spoof + // case fail for the real reason: transcript prose currently masquerades as + // the active bottom footer, while the genuine queue footer remains green. + tabBehavior: inferCodexTabBehavior(screenBeforeWrite ?? ''), + screenBeforeWrite, + inputProfile: recordedCase.configClass === 'recorded-default-01491' + ? recordedIssuedProfile + : { + cliVersion: corpus.provider.cliVersion, + upstreamTag: corpus.provider.upstreamTag, + configClass: recordedCase.configClass, + configOverrides: [...recordedCase.configOverrides], + }, + } } - return undefined -} -function replayRecordedInput(recordedCase: RecordedPromptInputCase): string[] { - const input = new SubmittedPromptInput() - // WHY Stage 26 intentionally describes the context Stage 27 must consume - // before production declares it. Casting only the bound test call lets the - // pre-repair implementation compile and fail behaviorally; weakening the - // fixture or editing production types merely to make a red test compile would - // collapse the required tests-before-implementation boundary. - const consume = input.consume.bind(input) as RecordedConsumer - const submitted: string[] = [] - - for (const modalWrite of recordedCase.nonComposerWrites ?? []) { - submitted.push(...consume( - modalWrite, - contextFor(recordedCase, recordedCase.modal), - )) + function recordedScreenBeforeChunk( + recordedCase: RecordedPromptInputCase, + chunkIndex: number, + ): string[] | undefined { + const finalChunkIndex = recordedCase.inputChunks.length - 1 + if (chunkIndex === finalChunkIndex) return recordedCase.screenBeforeFinalWrite + + // WHY these are the only retained intermediate frame boundaries in the + // sanitized corpus. The Stage 25 recorder captured history search after + // Ctrl+R, so its popup is the pre-Ctrl+C surface. It captured the active-turn + // footer before typing the queued draft. Inventing frames for every ordinary + // character would turn provider recordings back into imagined fixtures. + if (recordedCase.id === 'modal-ctrl-c-preserves-draft' && chunkIndex === 2) { + return recordedCase.popup + } + if (recordedCase.id === 'active-footer-tab-queue' && chunkIndex === 0) { + return recordedCase.activeTurnFooter + } + return undefined } - recordedCase.inputChunks.forEach((chunk, chunkIndex) => { - submitted.push(...consume( - chunk, - contextFor( - recordedCase, - recordedScreenBeforeChunk(recordedCase, chunkIndex), - ), - )) - }) + function replayRecordedInput(recordedCase: RecordedPromptInputCase): string[] { + const input = new SubmittedPromptInput() + // WHY Stage 26 intentionally describes the context Stage 27 must consume + // before production declares it. Casting only the bound test call lets the + // pre-repair implementation compile and fail behaviorally; weakening the + // fixture or editing production types merely to make a red test compile would + // collapse the required tests-before-implementation boundary. + const consume = input.consume.bind(input) as RecordedConsumer + const submitted: string[] = [] + + for (const modalWrite of recordedCase.nonComposerWrites ?? []) { + submitted.push(...consume( + modalWrite, + contextFor(recordedCase, recordedCase.modal), + )) + } - return submitted -} + recordedCase.inputChunks.forEach((chunk, chunkIndex) => { + submitted.push(...consume( + chunk, + contextFor( + recordedCase, + recordedScreenBeforeChunk(recordedCase, chunkIndex), + ), + )) + }) -function caseById(id: typeof recordedCaseIds[number]): RecordedPromptInputCase { - const recordedCase = corpus.cases.find(value => value.id === id) - if (!recordedCase) throw new Error(`missing recorded case ${id}`) - return recordedCase -} + return submitted + } -function stableFrame( - recorded: RecordedStableFrame, - layout: { - layoutEpoch: number - providerLayoutEpoch: number - layoutStartGeneration?: number - rowPaintGeneration?: number - cursorPaintGeneration?: number - } = { - layoutEpoch: 0, - providerLayoutEpoch: 0, - }, -): StableTerminalFrame { - const { - rowPaintGeneration, - ...frameLayout - } = layout - return { - generation: recorded.generation, - ...frameLayout, - cols: recorded.cols, - cursor: recorded.cursor, - rows: recorded.rows.map(row => ({ - text: row.text, - cells: [...row.text], - isWrapped: row.isWrapped, - paintGeneration: rowPaintGeneration, - })), + function caseById(id: typeof recordedCaseIds[number] | typeof POPUP_ENTER_CASE_IDS[number]): RecordedPromptInputCase { + const recordedCase = corpus.cases.find(value => value.id === id) + if (!recordedCase) throw new Error(`missing recorded case ${id}`) + return recordedCase } -} -function frameFromRows(rows: readonly string[], generation = 1): StableTerminalFrame { - return { - generation, - layoutEpoch: 0, - providerLayoutEpoch: 0, - cols: Math.max(140, ...rows.map(row => [...row].length)), - cursor: { x: 0, y: 0 }, - rows: rows.map(text => ({ text, cells: [...text], isWrapped: false })), + function stableFrame( + recorded: RecordedStableFrame, + layout: { + layoutEpoch: number + providerLayoutEpoch: number + layoutStartGeneration?: number + rowPaintGeneration?: number + cursorPaintGeneration?: number + } = { + layoutEpoch: 0, + providerLayoutEpoch: 0, + }, + ): StableTerminalFrame { + const { + rowPaintGeneration, + ...frameLayout + } = layout + return { + generation: recorded.generation, + ...frameLayout, + cols: recorded.cols, + cursor: recorded.cursor, + rows: recorded.rows.map(row => ({ + text: row.text, + cells: [...row.text], + isWrapped: row.isWrapped, + paintGeneration: rowPaintGeneration, + })), + } } -} -function issuedEvidence(): PromptInputEvidence { - return new PromptInputEvidence(recordedIssuedProfile) -} + function frameFromRows(rows: readonly string[], generation = 1): StableTerminalFrame { + return { + generation, + layoutEpoch: 0, + providerLayoutEpoch: 0, + cols: Math.max(140, ...rows.map(row => [...row].length)), + cursor: { x: 0, y: 0 }, + rows: rows.map(text => ({ text, cells: [...text], isWrapped: false })), + } + } -function prepareRecordedProfile(mode = 'recorded-safe') { - return prepareCodex01491PromptInputProfile({ - binary: process.execPath, - cwd: process.cwd(), - baseArgs: [appServerFixturePath], - env: { - ...process.env, - CODEX_PROFILE_FIXTURE_MODE: mode, - }, - }) -} + function issuedEvidence(): PromptInputEvidence { + return new PromptInputEvidence(recordedIssuedProfile) + } -describe('recorded Codex 0.149.1 prompt-input contract', () => { - it('executes the complete sanitized catalog with pinned provenance', () => { - const fixtureIds = corpus.cases.map(recordedCase => recordedCase.id) - const catalogIds = [...catalog.matchAll(/^\| `([^`]+)` \|/gm)] - .map(match => match[1]) - .filter(id => id !== 'capability-6244eac-recorded') - - // WHY fixture files and catalog prose can drift independently. This turns - // the inventory into an executable boundary: adding, omitting, or merely - // documenting a case cannot inflate coverage without a replayed assertion. - expect(fixtureIds).toEqual(recordedCaseIds) - expect(catalogIds).toEqual(recordedCaseIds) - expect(corpus).toMatchObject({ - schemaVersion: 1, - sanitizerVersion: 1, - provider: { - cliVersion: 'codex-cli 0.149.1', - binarySha256: 'f0d8762236594359b60cfbe17f4c7e945a3ce8d1c91e74778838c968d250fb6c', - upstreamTag: 'rust-v0.149.1', - }, - terminal: { cols: 140, rows: 42 }, - source: { - kind: 'real-codex-tui-local-canned-responses', - fixtureSseSha256: '66658b7a1d9b0e3b234de932f552b946b8a005520888f24e001a024fb9a29e5b', + function prepareRecordedProfile(mode = 'recorded-safe') { + return prepareCodex01491PromptInputProfile({ + binary: process.execPath, + cwd: process.cwd(), + baseArgs: [appServerFixturePath], + env: { + ...process.env, + CODEX_PROFILE_FIXTURE_MODE: mode, + CODEX_PROFILE_FIXTURE_RECORDING: `./${spec.configReadRecordingFile}`, }, }) - }) + } - it.each(casesById)('retains independently checkable provenance for %s', ( - _caseId, - recordedCase, - ) => { - // WHY prompt expectations are trustworthy only while they remain tied to - // the private raw PTY, rollout, and request streams that produced them. The - // hashes permit local revalidation without committing those identity-rich - // sources, and the two independent provider outputs must agree exactly. - expect(recordedCase.sourceLabel).toMatch(/^recorded-source-[0-9a-f]{16}$/) - expect(recordedCase.rawPtySha256).toMatch(/^[0-9a-f]{64}$/) - expect(recordedCase.rolloutSha256).toMatch(/^[0-9a-f]{64}$/) - if (recordedCase.rawRequestSha256 === null) { - expect(recordedCase.expectedSubmission).toBe(false) - } else { - expect(recordedCase.rawRequestSha256).toMatch(/^[0-9a-f]{64}$/) - } - expect(recordedCase.requestUserText).toBe(recordedCase.durableUserText) - expect(recordedCase.durableUserText === null) - .toBe(!recordedCase.expectedSubmission) - const lowerLayerConfig = recordedCase.lowerLayerConfig ?? [] - if (recordedCase.configClass === 'recorded-default-01491') { - expect(recordedCase.configOverrides).toEqual([]) - expect(lowerLayerConfig).toEqual([]) - } else if (recordedCase.configClass === 'explicit-cli-override') { - expect(recordedCase.configOverrides.length).toBeGreaterThan(0) - expect(lowerLayerConfig).toEqual([]) - } else { - expect(lowerLayerConfig.length).toBeGreaterThan(0) - } - }) + describe(spec.title, () => { + beforeAll(async () => { + const preparation = await prepareRecordedProfile() + if (!preparation.ok) throw new Error('recorded config/read profile fixture was refused') + // The issued profile must name the version whose recording issued it. + expect(preparation.profile.cliVersion).toBe(spec.provider.cliVersion.replace(/^codex-cli /, '')) + expect(preparation.profile.upstreamTag).toBe(spec.provider.upstreamTag) + recordedIssuedProfile = preparation.profile + }) - it('pins the exact rust-v0.149.1 config precedence and conflict sources', () => { - expect(configSource).toMatchObject({ - schemaVersion: 1, - upstreamTag: 'rust-v0.149.1', - upstreamCommitSha: 'ff29a44391deccde0aba0f8390337d7f3c319ea4', - recordedCases: { - validLowerLayer: 'lower-layer-keymap-valid-control', - issuedOverridesConflict: 'lower-layer-keymap-issued-profile-conflict', - }, + it('executes the complete sanitized catalog with pinned provenance', () => { + const fixtureIds = corpus.cases.map(recordedCase => recordedCase.id) + const catalogIds = [...catalog.matchAll(/^\| `([^`]+)` \|/gm)] + .map(match => match[1]) + .filter(id => id !== 'capability-6244eac-recorded') + + // WHY fixture files and catalog prose can drift independently. This turns + // the inventory into an executable boundary: adding, omitting, or merely + // documenting a case cannot inflate coverage without a replayed assertion. + expect(fixtureIds).toEqual(spec.recordsPopupEnterCases + ? [...recordedCaseIds, ...POPUP_ENTER_CASE_IDS] + : recordedCaseIds) + expect(catalogIds).toEqual([...recordedCaseIds, ...POPUP_ENTER_CASE_IDS]) + expect(corpus).toMatchObject({ + schemaVersion: 1, + sanitizerVersion: 1, + provider: spec.provider, + terminal: { cols: 140, rows: 42 }, + source: { + kind: 'real-codex-tui-local-canned-responses', + fixtureSseSha256: '66658b7a1d9b0e3b234de932f552b946b8a005520888f24e001a024fb9a29e5b', + }, + }) }) - expect(configSource.files.map(file => [file.path, file.sha256])).toEqual([ - ['codex-rs/config/src/config_layer_source.rs', '6816bf7bd44b1f2799aae30331b77a7e8231ccacdc4cd3d44d6485f9e1118364'], - ['codex-rs/config/src/loader/mod.rs', '53d66dce1cd81de3d86610ff2a75aed7f9049609cbefcd3694590c0acfc7c404'], - ['codex-rs/config/src/overrides.rs', 'd10b2c943a709d28395cde201f1b084da4d303afe4ff698f91f59f518c1a9e13'], - ['codex-rs/config/src/merge.rs', 'a7628f0da10f7f7e770fce5160ecbf1ca846b7d9db4b8ed92c6c9cce22c7fb11'], - ['codex-rs/tui/src/keymap.rs', '709feecb708a16b66af8685f0028191efc23b6fac8f87db6a8d45df2440ff604'], - ]) - expect(configSource.files.every(file => file.coordinates.length > 0)).toBe(true) - }) - it.each(priorReplayCasesById)('matches or safely declines recorded input evidence for %s', ( - _caseId, - recordedCase, - ) => { - const submitted = replayRecordedInput(recordedCase) - - if (recordedCase.configClass === 'explicit-cli-override' || - !recordedCase.expectedSubmission) { - // WHY an actual provider submission does not entitle an unproven input - // profile to reconstruct it. A remap, Vim mode, unbound Enter, or popup - // Tab must produce no ownership evidence; a plausible wrong prompt can - // authorize a same-CWD sibling rollout and is worse than a safe miss. - expect(submitted).toEqual([]) - return - } + it.each(casesById)('retains independently checkable provenance for %s', ( + _caseId, + recordedCase, + ) => { + // WHY prompt expectations are trustworthy only while they remain tied to + // the private raw PTY, rollout, and request streams that produced them. The + // hashes permit local revalidation without committing those identity-rich + // sources, and the two independent provider outputs must agree exactly. + expect(recordedCase.sourceLabel).toMatch(/^recorded-source-[0-9a-f]{16}$/) + expect(recordedCase.rawPtySha256).toMatch(/^[0-9a-f]{64}$/) + expect(recordedCase.rolloutSha256).toMatch(/^[0-9a-f]{64}$/) + if (recordedCase.rawRequestSha256 === null) { + expect(recordedCase.expectedSubmission).toBe(false) + } else { + expect(recordedCase.rawRequestSha256).toMatch(/^[0-9a-f]{64}$/) + } + expect(recordedCase.requestUserText).toBe(recordedCase.durableUserText) + expect(recordedCase.durableUserText === null) + .toBe(!recordedCase.expectedSubmission) + const lowerLayerConfig = recordedCase.lowerLayerConfig ?? [] + if (recordedCase.configClass === 'recorded-default-01491') { + expect(recordedCase.configOverrides).toEqual([]) + expect(lowerLayerConfig).toEqual([]) + } else if (recordedCase.configClass === 'explicit-cli-override') { + expect(recordedCase.configOverrides.length).toBeGreaterThan(0) + expect(lowerLayerConfig).toEqual([]) + } else { + expect(lowerLayerConfig.length).toBeGreaterThan(0) + } + }) - const exact = [recordedCase.durableUserText] - if (unicodeBoundaryCases.has(recordedCase.id)) { - // WHY Stage 27 may either implement the exact recorded Unicode boundary - // or fail closed until a version-pinned segmenter exists. It may not emit - // the code-point/ASCII approximation that the provider never submitted. - expect([[], exact]).toContainEqual(submitted) - return - } + it(`pins the exact ${spec.provider.upstreamTag} config precedence and conflict sources`, () => { + expect(configSource).toMatchObject({ + schemaVersion: 1, + upstreamTag: spec.provider.upstreamTag, + upstreamCommitSha: spec.upstreamCommitSha, + recordedCases: { + validLowerLayer: 'lower-layer-keymap-valid-control', + issuedOverridesConflict: 'lower-layer-keymap-issued-profile-conflict', + }, + }) + expect(configSource.files.map(file => [file.path, file.sha256])).toEqual(spec.configSourceFiles) + expect(configSource.files.every(file => file.coordinates.length > 0)).toBe(true) + }) - // WHY default ASCII, multiline navigation, modal restoration, trust - // exclusion, and provider-proven Tab already have fully enumerated recorded - // semantics. Failing closed here would hide independent substrate defects - // behind one broad invalidation switch instead of repairing the boundary. - expect(submitted).toEqual(exact) - }) + it.each(priorReplayCasesById)('matches or safely declines recorded input evidence for %s', ( + _caseId, + recordedCase, + ) => { + const submitted = replayRecordedInput(recordedCase) + + if (recordedCase.configClass === 'explicit-cli-override' || + !recordedCase.expectedSubmission) { + // WHY an actual provider submission does not entitle an unproven input + // profile to reconstruct it. A remap, Vim mode, unbound Enter, or popup + // Tab must produce no ownership evidence; a plausible wrong prompt can + // authorize a same-CWD sibling rollout and is worse than a safe miss. + expect(submitted).toEqual([]) + return + } + + const exact = [recordedCase.durableUserText] + if (unicodeBoundaryCases.has(recordedCase.id)) { + // WHY Stage 27 may either implement the exact recorded Unicode boundary + // or fail closed until a version-pinned segmenter exists. It may not emit + // the code-point/ASCII approximation that the provider never submitted. + expect([[], exact]).toContainEqual(submitted) + return + } + + // WHY default ASCII, multiline navigation, modal restoration, trust + // exclusion, and provider-proven Tab already have fully enumerated recorded + // semantics. Failing closed here would hide independent substrate defects + // behind one broad invalidation switch instead of repairing the boundary. + expect(submitted).toEqual(exact) + }) - it('CH-04 does not treat an unchanged newer provider redraw as edit acknowledgement', () => { - const recordedCase = caseById('unchanged-redraw-after-edit') - const trace = recordedCase.editAcknowledgementTrace - if (!trace) throw new Error('missing recorded CH-04 acknowledgement trace') - - // WHY a generation proves only that some provider bytes were parsed. Here - // the working-status bytes were emitted before the edit reached Codex; the - // recorder schedules the real processes so that tiny interval is visible. - // Draft plus cursor are unchanged, so generation 36 cannot acknowledge the - // `_EDIT` suffix even though it is newer than the pre-write frame. - expect(trace.unchangedAfterEdit.generation) - .toBeGreaterThan(trace.beforeEdit.generation) - expect(trace.unchangedAfterEdit.rows.map(row => row.text)) - .toEqual(trace.beforeEdit.rows.map(row => row.text)) - expect(trace.unchangedAfterEdit.cursor).toEqual(trace.beforeEdit.cursor) - expect(trace.setupDurableUserText).toBe(trace.setupRequestUserText) - - const evidence = issuedEvidence() - evidence.consume(trace.baseDraft, { frame: null }) - evidence.consume(trace.edit, { frame: stableFrame(trace.beforeEdit) }) - expect(evidence.consume('\r', { - frame: stableFrame(trace.unchangedAfterEdit), - })).toEqual([]) - }) + it('CH-04 does not treat an unchanged newer provider redraw as edit acknowledgement', () => { + const recordedCase = caseById('unchanged-redraw-after-edit') + const trace = recordedCase.editAcknowledgementTrace + if (!trace) throw new Error('missing recorded CH-04 acknowledgement trace') + + // WHY a generation proves only that some provider bytes were parsed. Here + // the working-status bytes were emitted before the edit reached Codex; the + // recorder schedules the real processes so that tiny interval is visible. + // Draft plus cursor are unchanged, so generation 36 cannot acknowledge the + // `_EDIT` suffix even though it is newer than the pre-write frame. + expect(trace.unchangedAfterEdit.generation) + .toBeGreaterThan(trace.beforeEdit.generation) + expect(trace.unchangedAfterEdit.rows.map(row => row.text)) + .toEqual(trace.beforeEdit.rows.map(row => row.text)) + expect(trace.unchangedAfterEdit.cursor).toEqual(trace.beforeEdit.cursor) + expect(trace.setupDurableUserText).toBe(trace.setupRequestUserText) + + const evidence = issuedEvidence() + evidence.consume(trace.baseDraft, { frame: null }) + evidence.consume(trace.edit, { frame: stableFrame(trace.beforeEdit) }) + expect(evidence.consume('\r', { + frame: stableFrame(trace.unchangedAfterEdit), + })).toEqual([]) + }) - it('CH-04 accepts the durable value only after the provider paints the edit', () => { - const recordedCase = caseById('unchanged-redraw-after-edit') - const trace = recordedCase.editAcknowledgementTrace - if (!trace) throw new Error('missing recorded CH-04 acknowledgement trace') - - expect(trace.paintedEdit.generation) - .toBeGreaterThan(trace.unchangedAfterEdit.generation) - const evidence = issuedEvidence() - evidence.consume(trace.baseDraft, { frame: null }) - evidence.consume(trace.edit, { frame: stableFrame(trace.beforeEdit) }) - expect(evidence.consume('\r', { - frame: stableFrame(trace.paintedEdit), - })).toEqual([recordedCase.durableUserText]) - }) + it('CH-04 accepts the durable value only after the provider paints the edit', () => { + const recordedCase = caseById('unchanged-redraw-after-edit') + const trace = recordedCase.editAcknowledgementTrace + if (!trace) throw new Error('missing recorded CH-04 acknowledgement trace') + + expect(trace.paintedEdit.generation) + .toBeGreaterThan(trace.unchangedAfterEdit.generation) + const evidence = issuedEvidence() + evidence.consume(trace.baseDraft, { frame: null }) + evidence.consume(trace.edit, { frame: stableFrame(trace.beforeEdit) }) + expect(evidence.consume('\r', { + frame: stableFrame(trace.paintedEdit), + })).toEqual([recordedCase.durableUserText]) + }) - it('CH-09 rejects the recorded resized frame until Codex repaints its layout', () => { - const recordedCase = caseById('narrow-soft-wrap-resize-redraw') - const beforeTyping = recordedCase.beforeTypingFrame - const resize = recordedCase.resizeTrace - if (!beforeTyping || !resize) throw new Error('missing recorded CH-04 frames') - - // WHY the equal generation and byte count are the causal boundary. xterm - // has adopted 92 columns, but Codex has not acknowledged that layout: its - // old 52-column two-row paint remains byte-for-byte on screen. Interpreting - // those rows with the new width manufactures a newline the durable prompt - // and request prove never existed. - expect(resize.beforeProviderRedraw.generation).toBe(resize.narrow.generation) - expect(resize.rawChunkCountBeforeProviderRedraw) - .toBe(resize.rawChunkCountBeforeResize) - expect(resize.beforeProviderRedraw.rows.map(row => row.text)) - .toEqual(resize.narrow.rows.map(row => row.text)) - - const evidence = issuedEvidence() - expect(evidence.consume(recordedCase.inputChunks[0]!, { - frame: stableFrame(beforeTyping), - })).toEqual([]) - expect(evidence.consume('\r', { - // WHY the recorded generation and raw-chunk count stayed unchanged - // across this resize. Epoch 1 is xterm's new geometry; epoch 0 is the - // latest geometry Codex had actually painted at this exact boundary. - frame: stableFrame(resize.beforeProviderRedraw, { - layoutEpoch: 1, - providerLayoutEpoch: 0, - }), - })).toEqual([]) - }) + it('CH-09 rejects the recorded resized frame until Codex repaints its layout', () => { + const recordedCase = caseById('narrow-soft-wrap-resize-redraw') + const beforeTyping = recordedCase.beforeTypingFrame + const resize = recordedCase.resizeTrace + if (!beforeTyping || !resize) throw new Error('missing recorded CH-04 frames') + + // WHY the equal generation and byte count are the causal boundary. xterm + // has adopted 92 columns, but Codex has not acknowledged that layout: its + // old 52-column two-row paint remains byte-for-byte on screen. Interpreting + // those rows with the new width manufactures a newline the durable prompt + // and request prove never existed. + expect(resize.beforeProviderRedraw.generation).toBe(resize.narrow.generation) + expect(resize.rawChunkCountBeforeProviderRedraw) + .toBe(resize.rawChunkCountBeforeResize) + expect(resize.beforeProviderRedraw.rows.map(row => row.text)) + .toEqual(resize.narrow.rows.map(row => row.text)) + + const evidence = issuedEvidence() + expect(evidence.consume(recordedCase.inputChunks[0]!, { + frame: stableFrame(beforeTyping), + })).toEqual([]) + expect(evidence.consume('\r', { + // WHY the recorded generation and raw-chunk count stayed unchanged + // across this resize. Epoch 1 is xterm's new geometry; epoch 0 is the + // latest geometry Codex had actually painted at this exact boundary. + frame: stableFrame(resize.beforeProviderRedraw, { + layoutEpoch: 1, + providerLayoutEpoch: 0, + }), + })).toEqual([]) + }) - it('Stage 38 rejects a post-resize status generation that leaves stale composer geometry', () => { - const resizeCase = caseById('narrow-soft-wrap-resize-redraw') - const statusCase = caseById('unchanged-redraw-after-edit') - const beforeTyping = resizeCase.beforeTypingFrame - const resize = resizeCase.resizeTrace - const status = statusCase.editAcknowledgementTrace - if (!beforeTyping || !resize || !status) { - throw new Error('missing recorded Stage 38 terminal boundaries') - } + it('Stage 38 rejects a post-resize status generation that leaves stale composer geometry', () => { + const resizeCase = caseById('narrow-soft-wrap-resize-redraw') + const statusCase = caseById('unchanged-redraw-after-edit') + const beforeTyping = resizeCase.beforeTypingFrame + const resize = resizeCase.resizeTrace + const status = statusCase.editAcknowledgementTrace + if (!beforeTyping || !resize || !status) { + throw new Error('missing recorded Stage 38 terminal boundaries') + } + + // WHY this composes two real recordings without inventing any composer + // content. The resize fixture supplies the exact stale xterm-reflowed rows + // whose newline disagrees with both its durable role-user item and request. + // The unrelated-redraw fixture supplies the independently observed + // scheduler fact: two later provider chunks/generations can repaint status + // while draft rows and logical cursor remain byte-identical. Replaying only + // that observed generation delta over the recorded stale rows models the + // coarse `providerLayoutEpoch` transition under review; no prompt atom is + // synthesized or changed. + expect(status.unchangedComposerRevision).toBe(true) + expect(status.unchangedAfterEdit.rows.map(row => row.text)) + .toEqual(status.beforeEdit.rows.map(row => row.text)) + expect(status.unchangedAfterEdit.cursor).toEqual(status.beforeEdit.cursor) + const statusGenerationDelta = status.unchangedAfterEdit.generation - + status.beforeEdit.generation + const statusChunkDelta = status.rawChunkCountAtUnchangedRedraw - + status.rawChunkCountBeforeEdit + expect(statusGenerationDelta).toBeGreaterThan(0) + expect(statusChunkDelta).toBe(statusGenerationDelta) + + const staleAfterStatus: RecordedStableFrame = { + ...resize.beforeProviderRedraw, + generation: resize.beforeProviderRedraw.generation + statusGenerationDelta, + } + expect(staleAfterStatus.rows.map(row => row.text)) + .toEqual(resize.beforeProviderRedraw.rows.map(row => row.text)) + + const evidence = issuedEvidence() + evidence.consume(resizeCase.inputChunks[0]!, { + frame: stableFrame(beforeTyping), + }) + expect(evidence.consume('\r', { + // A status-only chunk currently advances providerLayoutEpoch to 1 even + // though no Codex composer row was painted for layout epoch 1. Ownership + // must remain empty until the genuine recorded repaint used by the next + // green control, not accept the stale two-row geometry as a logical LF. + frame: stableFrame(staleAfterStatus, { + layoutEpoch: 1, + providerLayoutEpoch: 1, + // WHY the status chunks are real but their recording did not contain + // HeadlessTerminal's new provider-neutral paint metadata. Projecting + // the observed unchanged draft/cursor revision onto the resize fence + // says exactly what was captured: those physical cells remain owned by + // the layout-start generation even though the coarse epoch advanced. + layoutStartGeneration: resize.beforeProviderRedraw.generation, + rowPaintGeneration: resize.beforeProviderRedraw.generation, + cursorPaintGeneration: resize.beforeProviderRedraw.generation, + }), + })).toEqual([]) + }) - // WHY this composes two real recordings without inventing any composer - // content. The resize fixture supplies the exact stale xterm-reflowed rows - // whose newline disagrees with both its durable role-user item and request. - // The unrelated-redraw fixture supplies the independently observed - // scheduler fact: two later provider chunks/generations can repaint status - // while draft rows and logical cursor remain byte-identical. Replaying only - // that observed generation delta over the recorded stale rows models the - // coarse `providerLayoutEpoch` transition under review; no prompt atom is - // synthesized or changed. - expect(status.unchangedComposerRevision).toBe(true) - expect(status.unchangedAfterEdit.rows.map(row => row.text)) - .toEqual(status.beforeEdit.rows.map(row => row.text)) - expect(status.unchangedAfterEdit.cursor).toEqual(status.beforeEdit.cursor) - const statusGenerationDelta = status.unchangedAfterEdit.generation - - status.beforeEdit.generation - const statusChunkDelta = status.rawChunkCountAtUnchangedRedraw - - status.rawChunkCountBeforeEdit - expect(statusGenerationDelta).toBeGreaterThan(0) - expect(statusChunkDelta).toBe(statusGenerationDelta) - - const staleAfterStatus: RecordedStableFrame = { - ...resize.beforeProviderRedraw, - generation: resize.beforeProviderRedraw.generation + statusGenerationDelta, - } - expect(staleAfterStatus.rows.map(row => row.text)) - .toEqual(resize.beforeProviderRedraw.rows.map(row => row.text)) + it('CH-09 accepts the same durable prompt after the recorded provider redraw', () => { + const recordedCase = caseById('narrow-soft-wrap-resize-redraw') + const beforeTyping = recordedCase.beforeTypingFrame + const resize = recordedCase.resizeTrace + if (!beforeTyping || !resize) throw new Error('missing recorded CH-04 frames') + + expect(resize.afterProviderRedraw.generation) + .toBeGreaterThan(resize.beforeProviderRedraw.generation) + const evidence = issuedEvidence() + evidence.consume(recordedCase.inputChunks[0]!, { + frame: stableFrame(beforeTyping), + }) + expect(evidence.consume('\r', { + frame: stableFrame(resize.afterProviderRedraw, { + layoutEpoch: 1, + providerLayoutEpoch: 1, + // WHY unlike the Stage 38 boundary, this real frame changed both the + // composer geometry and logical cursor after SIGWINCH. The projection + // records only those captured facts; it does not invent draft text. + layoutStartGeneration: resize.beforeProviderRedraw.generation, + rowPaintGeneration: resize.afterProviderRedraw.generation, + cursorPaintGeneration: resize.afterProviderRedraw.generation, + }), + })).toEqual([recordedCase.durableUserText]) + }) - const evidence = issuedEvidence() - evidence.consume(resizeCase.inputChunks[0]!, { - frame: stableFrame(beforeTyping), + it('CH-10 keeps modal sentinel prose inside an ordinary submitted draft', () => { + const recordedCase = caseById('ordinary-modal-sentinel-draft') + if (!recordedCase.screenBeforeFinalWrite) throw new Error('missing CH-05 screen') + const evidence = issuedEvidence() + evidence.consume(recordedCase.inputChunks[0]!, { frame: null }) + expect(evidence.consume('\r', { + frame: frameFromRows(recordedCase.screenBeforeFinalWrite), + })).toEqual([recordedCase.durableUserText]) }) - expect(evidence.consume('\r', { - // A status-only chunk currently advances providerLayoutEpoch to 1 even - // though no Codex composer row was painted for layout epoch 1. Ownership - // must remain empty until the genuine recorded repaint used by the next - // green control, not accept the stale two-row geometry as a logical LF. - frame: stableFrame(staleAfterStatus, { - layoutEpoch: 1, - providerLayoutEpoch: 1, - // WHY the status chunks are real but their recording did not contain - // HeadlessTerminal's new provider-neutral paint metadata. Projecting - // the observed unchanged draft/cursor revision onto the resize fence - // says exactly what was captured: those physical cells remain owned by - // the layout-start generation even though the coarse epoch advanced. - layoutStartGeneration: resize.beforeProviderRedraw.generation, - rowPaintGeneration: resize.beforeProviderRedraw.generation, - cursorPaintGeneration: resize.beforeProviderRedraw.generation, - }), - })).toEqual([]) - }) - it('CH-09 accepts the same durable prompt after the recorded provider redraw', () => { - const recordedCase = caseById('narrow-soft-wrap-resize-redraw') - const beforeTyping = recordedCase.beforeTypingFrame - const resize = recordedCase.resizeTrace - if (!beforeTyping || !resize) throw new Error('missing recorded CH-04 frames') - - expect(resize.afterProviderRedraw.generation) - .toBeGreaterThan(resize.beforeProviderRedraw.generation) - const evidence = issuedEvidence() - evidence.consume(recordedCase.inputChunks[0]!, { - frame: stableFrame(beforeTyping), + it('CH-10 treats Vim: Insert in the recorded cwd footer as ordinary text', () => { + const recordedCase = caseById('ordinary-vim-sentinel-cwd') + if (!recordedCase.screenBeforeFinalWrite) throw new Error('missing CH-09 screen') + const evidence = issuedEvidence() + evidence.consume(recordedCase.inputChunks[0]!, { frame: null }) + expect(evidence.consume('\r', { + frame: frameFromRows(recordedCase.screenBeforeFinalWrite), + })).toEqual([recordedCase.durableUserText]) }) - expect(evidence.consume('\r', { - frame: stableFrame(resize.afterProviderRedraw, { - layoutEpoch: 1, - providerLayoutEpoch: 1, - // WHY unlike the Stage 38 boundary, this real frame changed both the - // composer geometry and logical cursor after SIGWINCH. The projection - // records only those captured facts; it does not invent draft text. - layoutStartGeneration: resize.beforeProviderRedraw.generation, - rowPaintGeneration: resize.afterProviderRedraw.generation, - cursorPaintGeneration: resize.afterProviderRedraw.generation, - }), - })).toEqual([recordedCase.durableUserText]) - }) - it('CH-10 keeps modal sentinel prose inside an ordinary submitted draft', () => { - const recordedCase = caseById('ordinary-modal-sentinel-draft') - if (!recordedCase.screenBeforeFinalWrite) throw new Error('missing CH-05 screen') - const evidence = issuedEvidence() - evidence.consume(recordedCase.inputChunks[0]!, { frame: null }) - expect(evidence.consume('\r', { - frame: frameFromRows(recordedCase.screenBeforeFinalWrite), - })).toEqual([recordedCase.durableUserText]) - }) + it('CH-10 still rejects the genuine recorded trust modal and Vim status atom', () => { + const trust = caseById('trust-action-then-submit') + const vim = caseById('vim-normal-default') + if (!trust.modal || !vim.screenBeforeFinalWrite) { + throw new Error('missing recorded CH-10 negative controls') + } + + expect(classifyCodex01491ComposerSurface(frameFromRows(trust.modal))) + .toEqual({ kind: 'non-composer-modal' }) + expect(classifyCodex01491ComposerSurface( + frameFromRows(vim.screenBeforeFinalWrite), + )).toEqual({ kind: 'unknown' }) + }) - it('CH-10 treats Vim: Insert in the recorded cwd footer as ordinary text', () => { - const recordedCase = caseById('ordinary-vim-sentinel-cwd') - if (!recordedCase.screenBeforeFinalWrite) throw new Error('missing CH-09 screen') - const evidence = issuedEvidence() - evidence.consume(recordedCase.inputChunks[0]!, { frame: null }) - expect(evidence.consume('\r', { - frame: frameFromRows(recordedCase.screenBeforeFinalWrite), - })).toEqual([recordedCase.durableUserText]) - }) + it('classifies the recorded skill popup as a popup, never as a composer', () => { + // #63: 0.157.1 paints this popup ABOVE the composer and no longer uses + // the 0.149.1 footer string. Read as an idle composer, Enter in the popup + // (which inserts a completion) would have produced submission evidence for + // a prompt Codex never sent. The replayed case alone declines either way + // (Tab never submits), so the surface itself is pinned here. + const popupCase = caseById('tab-footer-spoof-skill-popup') + if (!popupCase.popup) throw new Error('missing recorded popup frame') + expect(classifyCodex01491ComposerSurface(frameFromRows(popupCase.popup))) + .toEqual({ kind: 'completion-popup' }) + }) - it('CH-10 still rejects the genuine recorded trust modal and Vim status atom', () => { - const trust = caseById('trust-action-then-submit') - const vim = caseById('vim-normal-default') - if (!trust.modal || !vim.screenBeforeFinalWrite) { - throw new Error('missing recorded CH-10 negative controls') - } + it.runIf(spec.recordsPopupEnterCases).each(POPUP_ENTER_CASE_IDS)('never counts Enter in the recorded %s frame as a submission', caseId => { + // Review a of codex-headless#69: these popups paint above the composer + // with no reliable hint row, and Enter selects the popup item. Codex + // submitted nothing (the recording asserts no rollout user item and no + // request), so the issued evidence must emit nothing either. + const recordedCase = caseById(caseId) + if (!recordedCase.screenBeforeFinalWrite) throw new Error(`missing recorded frame for ${caseId}`) + expect(recordedCase.expectedSubmission).toBe(false) + expect(classifyCodex01491ComposerSurface(frameFromRows(recordedCase.screenBeforeFinalWrite))) + .toEqual({ kind: 'completion-popup' }) + const evidence = issuedEvidence() + evidence.consume(recordedCase.inputChunks[0]!, { frame: null }) + expect(evidence.consume('\r', { frame: frameFromRows(recordedCase.screenBeforeFinalWrite) })).toEqual([]) + }) - expect(classifyCodex01491ComposerSurface(frameFromRows(trust.modal))) - .toEqual({ kind: 'non-composer-modal' }) - expect(classifyCodex01491ComposerSurface( - frameFromRows(vim.screenBeforeFinalWrite), - )).toEqual({ kind: 'unknown' }) - }) + it('reads the recorded idle footer rows as a composer that does not queue with Tab', () => { + // Review a: the fullscreen `? for shortcuts` instructional row was never + // asserted. The active-turn frame of the queue case shows it (fullscreen) + // or the one-row footer (inline); either way it is a composer, and only + // `tab to queue` rows make Tab queue. + const frame = caseById('active-footer-tab-queue').activeTurnFooter + if (!frame) throw new Error('missing recorded active-turn footer') + expect(classifyCodex01491ComposerSurface(frameFromRows(frame))) + .toMatchObject({ kind: 'primary-composer', queueWithTab: false }) + }) - it('CH-05 refuses profile issuance for the recorded lower-layer conflict', async () => { - const control = caseById('lower-layer-keymap-valid-control') - const conflict = caseById('lower-layer-keymap-issued-profile-conflict') - expect(control.startupOutcome).toBe('composer-ready') - expect(control.requestCountDelta).toBe(0) - expect(conflict.startupOutcome).toBe('rejected-before-composer') - expect(conflict.exitOutcome?.exitCode).toBe(1) - expect(conflict.requestCountDelta).toBe(0) - - // WHY the fixture server wraps the recorded config/read projection and - // adds only the exact non-null lower binding proved by the paired live TUI - // outcomes above. The issuer must inspect effective routing and refuse - // before Agent Code appends arguments that make Codex exit at startup. - await expect(prepareRecordedProfile('conflicting-binding')).resolves.toEqual({ - ok: false, - reason: 'effective-config-unverified', + it('CH-05 refuses profile issuance for the recorded lower-layer conflict', async () => { + const control = caseById('lower-layer-keymap-valid-control') + const conflict = caseById('lower-layer-keymap-issued-profile-conflict') + expect(control.startupOutcome).toBe('composer-ready') + expect(control.requestCountDelta).toBe(0) + expect(conflict.startupOutcome).toBe('rejected-before-composer') + expect(conflict.exitOutcome?.exitCode).toBe(1) + expect(conflict.requestCountDelta).toBe(0) + + // WHY the fixture server wraps the recorded config/read projection and + // adds only the exact non-null lower binding proved by the paired live TUI + // outcomes above. The issuer must inspect effective routing and refuse + // before Agent Code appends arguments that make Codex exit at startup. + await expect(prepareRecordedProfile('conflicting-binding')).resolves.toEqual({ + ok: false, + reason: 'effective-config-unverified', + }) }) }) +} + +defineRecordedCorpusSuite({ + title: 'recorded Codex 0.149.1 prompt-input contract', + corpusFile: 'codex-01491-recorded.json', + configSourceFile: 'codex-01491-config-source.json', + configReadRecordingFile: 'codex-01491-config-read-recorded.json', + provider: { + cliVersion: 'codex-cli 0.149.1', + binarySha256: 'f0d8762236594359b60cfbe17f4c7e945a3ce8d1c91e74778838c968d250fb6c', + upstreamTag: 'rust-v0.149.1', + }, + upstreamCommitSha: 'ff29a44391deccde0aba0f8390337d7f3c319ea4', + recordsPopupEnterCases: false, + configSourceFiles: [ + ['codex-rs/config/src/config_layer_source.rs', '6816bf7bd44b1f2799aae30331b77a7e8231ccacdc4cd3d44d6485f9e1118364'], + ['codex-rs/config/src/loader/mod.rs', '53d66dce1cd81de3d86610ff2a75aed7f9049609cbefcd3694590c0acfc7c404'], + ['codex-rs/config/src/overrides.rs', 'd10b2c943a709d28395cde201f1b084da4d303afe4ff698f91f59f518c1a9e13'], + ['codex-rs/config/src/merge.rs', 'a7628f0da10f7f7e770fce5160ecbf1ca846b7d9db4b8ed92c6c9cce22c7fb11'], + ['codex-rs/tui/src/keymap.rs', '709feecb708a16b66af8685f0028191efc23b6fac8f87db6a8d45df2440ff604'], + ], }) + +// Recorded inline (--no-alt-screen, like the 0.149.1 corpus) and again in +// fullscreen, which is 0.157's default and how Agent Code launches Codex. +for (const [layout, corpusFile] of [ + ['inline', 'codex-01571-recorded.json'], + ['fullscreen', 'codex-01571-fullscreen-recorded.json'], +] as const) { + defineRecordedCorpusSuite({ + title: `recorded Codex 0.157.1 prompt-input contract (${layout})`, + corpusFile, + configSourceFile: 'codex-01571-config-source.json', + configReadRecordingFile: 'codex-01571-config-read-recorded.json', + provider: { + cliVersion: 'codex-cli 0.157.1', + binarySha256: '27ceb5f9b957b43a519efe4eaa3816a0bffb0a531a2c89af18840c0a3c016a7d', + upstreamTag: 'rust-v0.157.1', + }, + upstreamCommitSha: '36650394c5b38c2990ccf2a3457165ca3e9d9726', + recordsPopupEnterCases: true, + configSourceFiles: [ + ['codex-rs/config/src/config_layer_source.rs', '6816bf7bd44b1f2799aae30331b77a7e8231ccacdc4cd3d44d6485f9e1118364'], + ['codex-rs/config/src/loader/mod.rs', '0e3131c8186b391ecb425620bee2a8649d10364d062eee21a15d1cbe48dbbb0d'], + ['codex-rs/config/src/overrides.rs', 'dbac7a6f68de31e74c11e9671fe0b1f68f8783c68cff5a9ba26348e0b5f86c0a'], + ['codex-rs/config/src/merge.rs', 'aca71a56a375d83e79009b59b4c1421e70790d625694b4601cd13f182003817f'], + ['codex-rs/tui/src/keymap.rs', '006feab4bc833464cce59c11e0d09dae82b7e8f3f64e2885b3cc9e12b56e02d8'], + ], + }) +} diff --git a/src/transcript/prompt-input/Codex01491ComposerSurface.popups.test.ts b/src/transcript/prompt-input/Codex01491ComposerSurface.popups.test.ts new file mode 100644 index 0000000..98e6c67 --- /dev/null +++ b/src/transcript/prompt-input/Codex01491ComposerSurface.popups.test.ts @@ -0,0 +1,79 @@ +import { describe, expect, it } from 'vitest' + +import type { StableTerminalFrame } from '../../terminal/HeadlessTerminal.js' +import { classifyCodex01491ComposerSurface } from './Codex01491ComposerSurface.js' + +// Review b of codex-headless#69. These are the rows of upstream's own insta +// snapshot `chat_composer__status_surface__tests__slash_popup_footer_wide` +// (rust-v0.157.1), verbatim apart from trailing padding: a command popup +// painted ABOVE the composer, with no hint row, over a status line that has +// the accepted footer shape. Enter here dispatches the selected `/memories` +// (slash_input.rs), so the `/m` draft must never read as a composer about to +// submit it. The recorded 0.157.1 equivalents (slash and file popups) are in +// SubmittedPromptInput.recorded.test.ts. +function frame(rows: string[]): StableTerminalFrame { + return { + generation: 1, + layoutEpoch: 0, + providerLayoutEpoch: 0, + cols: 80, + cursor: { x: 4, y: 5 }, + rows: rows.map(text => ({ text, cells: [...text], isWrapped: false })), + } +} + +describe('Codex 0.157 popups above the composer', () => { + it('classifies the upstream slash popup snapshot as a popup, not a composer', () => { + expect(classifyCodex01491ComposerSurface(frame([ + ' /model choose what model and reasoning effort to use', + '› /memories configure memory use and generation', + ' /mention mention a file', + ' /mcp list configured MCP tools; use /mcp verbose for details', + '', + '› /m', + '', + ' model · high · fast', + ]))).toEqual({ kind: 'completion-popup' }) + }) + + it('still reads an ordinary draft over the same status line as a composer', () => { + expect(classifyCodex01491ComposerSurface(frame([ + '› explain the memory settings', + '', + ' model · high · fast', + ]))).toMatchObject({ kind: 'primary-composer', draftText: 'explain the memory settings' }) + }) + + it.each(['@', 'look at @', '$', 'use $', '/'])('declines a draft that ends in a bare popup sigil: %j', draft => { + // The popup opens on the sigil alone, before any character follows it. + expect(classifyCodex01491ComposerSurface(frame([`› ${draft}`, '', ' model · high · fast']))) + .toEqual({ kind: 'completion-popup' }) + }) + // Review c of #69 (F2): the hint-row layer was only ever exercised with a + // draft the draft rule catches too, so deleting it kept every test green. + // Here the draft (`/` has been accepted into a plain word) matches no + // sigil, and only the painted "enter insert · esc close" row proves the + // popup still owns Enter. + it('declines a composer under a painted completion hint even when the draft has no sigil', () => { + expect(classifyCodex01491ComposerSurface(frame([ + ' README.md', + ' enter insert · esc close', + '', + '› look at the readme', + '', + ' model · high · fast', + ]))).toEqual({ kind: 'completion-popup' }) + }) + + // Review c of #69 (F3): the fullscreen pair accepts only the two exact + // instructional rows under the status line. Any other second row is an + // unrecorded bottom pane and must not read as a composer. + it('does not accept an unrecorded second footer row under the status line', () => { + expect(classifyCodex01491ComposerSurface(frame([ + '› explain the memory settings', + '', + ' model · high · fast', + ' press enter to approve', + ]))).toEqual({ kind: 'unknown' }) + }) +}) diff --git a/src/transcript/prompt-input/Codex01491ComposerSurface.test.ts b/src/transcript/prompt-input/Codex01491ComposerSurface.test.ts new file mode 100644 index 0000000..eb9651a --- /dev/null +++ b/src/transcript/prompt-input/Codex01491ComposerSurface.test.ts @@ -0,0 +1,51 @@ +import { describe, expect, it } from 'vitest' + +import type { StableTerminalFrame } from '../../terminal/HeadlessTerminal.js' +import { classifyCodex01491ComposerSurface } from './Codex01491ComposerSurface.js' + +// Review c of #67: the 0.156+ trust-hint anchor in this surface survived the +// whole suite when broken. With it broken, the new trust dialog reads as +// something other than a modal to the prompt-input surface, so these pin it, +// with the hint as Codex paints it (last row) and wrapped once (the Windows +// variant below 46 columns). Rows are the 80-column 0.157.1 dialog from +// testing/fixtures/trust-dialog-0157, below its hard-wrapped path. +function frame(rows: string[]): StableTerminalFrame { + return { + generation: 1, + layoutEpoch: 0, + providerLayoutEpoch: 0, + cols: 80, + cursor: { x: 0, y: 0 }, + rows: rows.map(text => ({ text, cells: [...text], isWrapped: false })), + } +} + +const DIALOG = [ + ' Folder access', + ' /private/tmp/claude-501/-Users-fixture-user-Desktop-Development-agent-code/d', + ' 0000000-0000-0000-0000-000000000000/scratchpad/rec/untrusted-ABCD', + '', + ' Trust this folder? Codex can read, edit, and run files here, subject to your', + ' permission settings. Folder settings can run code automatically, even', + ' without a model request. Continue only if you trust these files. Your trust', + ' decision will be saved.', + '', + '› 1. Trust and continue', + ' 2. Back to Agent Command Center', + '', +] + +describe('classifyCodex01491ComposerSurface on the 0.156+ trust dialog', () => { + it('treats the dialog as a modal, not a composer', () => { + expect(classifyCodex01491ComposerSurface(frame([...DIALOG, ' enter continue · esc back']))) + .toEqual({ kind: 'non-composer-modal' }) + }) + + it('treats the dialog as a modal when the Windows hint wraps', () => { + expect(classifyCodex01491ComposerSurface(frame([ + ...DIALOG, + ' enter continue and create sandbox ·', + ' esc back', + ]))).toEqual({ kind: 'non-composer-modal' }) + }) +}) diff --git a/src/transcript/prompt-input/Codex01491ComposerSurface.ts b/src/transcript/prompt-input/Codex01491ComposerSurface.ts index 46f5e8c..f641aeb 100644 --- a/src/transcript/prompt-input/Codex01491ComposerSurface.ts +++ b/src/transcript/prompt-input/Codex01491ComposerSurface.ts @@ -14,8 +14,39 @@ export type Codex01491ComposerSurface = const PLACEHOLDER = 'Ask Codex to do anything' const HISTORY_FOOTER = /^\s{2}reverse-i-search:/i const COMPLETION_FOOTER = /^\s{2}Press enter to insert or esc to close\s*$/i +// WHY a two-row footer shape (#63). In 0.157.1's fullscreen mode, the default +// and how Agent Code launches Codex, the bottom pane has room for BOTH the +// status line and footer.rs's instructional row under it: "? for shortcuts" +// while idle, "tab to queue message" (or, when narrow, "tab to queue") when +// Tab would queue the draft. Both are recorded in +// codex-01571-fullscreen-recorded.json. Only these exact instructional rows +// are accepted (an optional right-aligned context percentage allowed). +// footer.rs has more variants (agent hints, the collaboration-mode indicator, +// the cycle hint), and any of those makes the pane `unknown`, which declines +// evidence rather than guessing. +const FULLSCREEN_SHORTCUTS_HINT = /^ \? for shortcuts(?: {2,}\d+% context left)?$/u +const FULLSCREEN_QUEUE_HINT = /^ tab to queue(?: message)?(?: {2,}\d+% context left)?$/u + +// WHY a second popup shape (#63): 0.157.1 moved the skill/mention popup ABOVE +// the composer (skill_popup.rs at rust-v0.157.1; recorded in the 0.157.1 +// corpus) and dropped the 0.149.1 footer string entirely: the popup's hint +// row, then a blank row, then the composer. Below the composer the pane +// then looks exactly like an idle composer with a draft, so without this check +// Enter in the popup (which INSERTS a completion) would be recorded as +// submitting the draft, the plausible wrong prompt this parser exists to +// refuse. A transcript line that happens to read like the hint just above the +// composer only makes us decline, which is the safe direction. +const COMPLETION_HINT_ABOVE_COMPOSER = /^\s{2}enter(?:\/tab)? insert · esc close(?: · .*)?$/i const QUEUE_FOOTER = /^ tab to queue(?: message)?\s+\d+% context left\s*$/i const IDLE_FOOTER = /^ \S.*\s·\s.+$/u +// The 0.156+ trust dialog's key hint (#63, codex-headless#65). Codex paints it +// as the dialog's LAST row, where a composer paints its status footer, and it +// also has the IDLE_FOOTER shape (" enter continue · esc quit"). Without this +// check the dialog was read as a composer whose "draft" is +// "1. Trust and continue". It is matched on the bottom row only (one wrap +// allowed: the Windows "… and create sandbox ·" hint wraps below 46 columns), +// so the same words typed into a draft are never mistaken for it. +const TRUST_HINT_FOOTER = /^\s*enter continue(?: and create sandbox)?\s*·\s*esc (?:quit|back)\s*$/i // Codex 0.149.1 renders Vim mode as a distinct right-hand status atom, with a // run of layout padding before the atom and no content after it. A cwd ending // in `/Vim: Insert` is part of the left status value and has neither boundary. @@ -43,8 +74,17 @@ export function classifyCodex01491ComposerSurface( const bottom = rows[lastNonBlank] ?? '' if (HISTORY_FOOTER.test(bottom)) return { kind: 'history-search' } if (COMPLETION_FOOTER.test(bottom)) return { kind: 'completion-popup' } + const aboveBottom = rows[lastNonBlank - 1] ?? '' + if (TRUST_HINT_FOOTER.test(bottom) || + (aboveBottom.trim() !== '' && TRUST_HINT_FOOTER.test(`${aboveBottom.trim()} ${bottom.trim()}`))) { + return { kind: 'non-composer-modal' } + } const composerRow = findComposerRow(rows, lastNonBlank) + if (composerRow >= 2 && (rows[composerRow - 1] ?? '').trim() === '' && + COMPLETION_HINT_ABOVE_COMPOSER.test(rows[composerRow - 2] ?? '')) { + return { kind: 'completion-popup' } + } if (composerRow >= 0) { const separatorRow = rows.findIndex((row, index) => index > composerRow && row.trim() === '', @@ -59,9 +99,17 @@ export function classifyCodex01491ComposerSurface( // owns them. A primary composer has exactly one anchored provider footer; // genuine trust/approval overlays have option rows plus their own footer // and therefore fall through to modal classification below. - if (footerRows.length === 1 && - (IDLE_FOOTER.test(bottom) || QUEUE_FOOTER.test(bottom))) { - if (VIM_STATUS_SUFFIX.test(bottom)) { + // One footer row (0.149.1, and 0.157.1 inline), or the 0.157.1 + // fullscreen pair: status line, then an exact instructional row. + const footer = footerRows.length === 1 && + (IDLE_FOOTER.test(bottom) || QUEUE_FOOTER.test(bottom)) + ? { status: bottom, queueWithTab: QUEUE_FOOTER.test(bottom) } + : footerRows.length === 2 && IDLE_FOOTER.test(footerRows[0]!) && + (FULLSCREEN_SHORTCUTS_HINT.test(footerRows[1]!) || FULLSCREEN_QUEUE_HINT.test(footerRows[1]!)) + ? { status: footerRows[0]!, queueWithTab: FULLSCREEN_QUEUE_HINT.test(footerRows[1]!) } + : null + if (footer) { + if (VIM_STATUS_SUFFIX.test(footer.status)) { // WHY `/vim` can change the live editor after launch even though the // issued profile forces a non-Vim startup. Only the provider's // right-separated footer atom proves that drift; the same words in a @@ -72,11 +120,12 @@ export function classifyCodex01491ComposerSurface( const draftRows = frame.rows.slice(composerRow, separatorRow) const draftText = extractDraftText(draftRows, frame.cols) if (draftText === null) return { kind: 'unknown' } + if (draftMayOpenPopup(draftText)) return { kind: 'completion-popup' } return { kind: 'primary-composer', draftText, - queueWithTab: QUEUE_FOOTER.test(bottom), + queueWithTab: footer.queueWithTab, } } } @@ -154,9 +203,39 @@ function findPreviousNonBlank(rows: readonly string[], from: number): number { } function isKnownNonComposerModal(text: string): boolean { + // The 0.156+ trust dialog is recognised structurally by TRUST_HINT_FOOTER + // before this point, since its hint is always the pane's last row. return /Do you trust the contents of this directory/i.test(text) || /Press enter to continue/i.test(text) || /Would you like to run the following command/i.test(text) || /Yes, and don't ask again/i.test(text) || /customize shortcuts with \/keymap/i.test(text) } + +// WHY decline on the DRAFT, not only on a visible popup (review a of +// codex-headless#69). Codex 0.157.1 paints every popup above the composer: +// the slash-command and file popups carry no hint row at all, and a short +// skill popup omits its hint. A frame with a popup open is therefore +// indistinguishable, from the pane alone, from an idle composer holding the +// same draft. Enter there selects or inserts the popup item (dispatches +// `/status`, inserts a file path) and submits nothing. Recorded in +// codex-01571-*recorded.json: `slash-popup-enter-selects-command` and +// `file-popup-enter-inserts-mention`. +// +// Codex opens these popups from the draft itself: a leading `/` for +// commands, and an `@` or `$` token for file, mention and skill search +// (chat_composer.rs, rust-v0.157.1). So a draft that could have one open never +// yields prompt evidence. This is deliberately fail-closed. A real prompt +// that starts with `/`, or mentions `$HOME` or `a@b`, becomes a safe miss +// (ownership falls back to the proxy path) rather than risking a false +// prompt, which could claim a sibling rollout. +// +// The draft rule, not the hint rows, is the primary guard (review c of #69): a +// popup can be open with ZERO rendered rows (a short pane, an empty query), so +// no row-shape check could ever be the only guard. +function draftMayOpenPopup(draft: string): boolean { + if (draft.trimStart().startsWith('/')) return true + // Any token that STARTS with the sigil, including the bare sigil itself: the + // popup opens on `@` / `$` before a single character follows it. + return /(?:^|\s)[@$]/u.test(draft) +} diff --git a/src/transcript/prompt-input/CodexPromptInputProfile.test.ts b/src/transcript/prompt-input/CodexPromptInputProfile.test.ts index a9ae588..0a99c2c 100644 --- a/src/transcript/prompt-input/CodexPromptInputProfile.test.ts +++ b/src/transcript/prompt-input/CodexPromptInputProfile.test.ts @@ -76,3 +76,22 @@ describe('Codex 0.149.1 prompt-input launch profile', () => { await expect(prepare(mode)).resolves.toEqual({ ok: false, reason }) }) }) + +// Review c of codex-headless#69 (F3): the safety property "a table of exact +// recorded versions, never a range" was enforced only by review; widening the +// table kept every test green. Pin it to exactly the versions that have a +// recorded corpus, each mapped to its own upstream tag. +describe('recorded prompt-input versions', () => { + it('issues profiles for exactly the recorded versions', async () => { + const { RECORDED_PROMPT_INPUT_VERSIONS } = await import('./CodexPromptInputProfile.js') + expect(RECORDED_PROMPT_INPUT_VERSIONS).toEqual({ + '0.149.1': 'rust-v0.149.1', + '0.157.1': 'rust-v0.157.1', + }) + const { existsSync } = await import('node:fs') + const { fileURLToPath } = await import('node:url') + for (const corpus of ['codex-01491-recorded.json', 'codex-01571-recorded.json', 'codex-01571-fullscreen-recorded.json']) { + expect(existsSync(fileURLToPath(new URL(`../../../testing/fixtures/prompt-input/${corpus}`, import.meta.url)))).toBe(true) + } + }) +}) diff --git a/src/transcript/prompt-input/CodexPromptInputProfile.ts b/src/transcript/prompt-input/CodexPromptInputProfile.ts index 8828f4d..cba8cd3 100644 --- a/src/transcript/prompt-input/CodexPromptInputProfile.ts +++ b/src/transcript/prompt-input/CodexPromptInputProfile.ts @@ -14,6 +14,28 @@ const CODEX_01491_PROMPT_INPUT_ARGS = Object.freeze( override, ]), ) +// WHY a table of EXACT recorded versions, never a range (#63). A profile is +// a claim that Agent Code can reconstruct what Codex will submit from the keys +// it forwards. That claim is only as good as the recorded corpus behind it +// (testing/fixtures/prompt-input): every row here has its own full live +// recording, taken with record-live-prompt-input.mts, plus a config/read +// recording and a per-tag audit of the upstream config precedence code. +// 0.157.1 was re-recorded against the same 16 cases on 2026-09-27. It agrees with +// 0.149.1 on every issued-profile case; the one difference, a Vim-default +// composer opening in Insert rather than Normal, sits outside the profile, +// which forces Vim off. A version that is not in this table has not been +// recorded, and it gets `unsupported-cli`. +export const RECORDED_PROMPT_INPUT_VERSIONS = Object.freeze({ + '0.149.1': 'rust-v0.149.1', + '0.157.1': 'rust-v0.157.1', +} as const) +type RecordedPromptInputVersion = keyof typeof RECORDED_PROMPT_INPUT_VERSIONS + +function isRecordedPromptInputVersion(value: unknown): value is RecordedPromptInputVersion { + return typeof value === 'string' && + Object.prototype.hasOwnProperty.call(RECORDED_PROMPT_INPUT_VERSIONS, value) +} + const PROBE_CLIENT_NAME = 'agent_code_prompt_profile_probe' const INITIALIZE_ID = 'agent-code-prompt-profile-initialize' const CONFIG_READ_ID = 'agent-code-prompt-profile-config-read' @@ -24,8 +46,8 @@ declare const CODEX_PROMPT_INPUT_PROFILE: unique symbol export type CodexPromptInputProfile = Readonly<{ profileVersion: 1 provider: 'codex' - cliVersion: '0.149.1' - upstreamTag: 'rust-v0.149.1' + cliVersion: RecordedPromptInputVersion + upstreamTag: (typeof RECORDED_PROMPT_INPUT_VERSIONS)[RecordedPromptInputVersion] submitKey: 'enter' queueKey: 'tab' vimMode: false @@ -67,7 +89,10 @@ export type CodexPromptInputProfilePreparation = * source of truth. `config/read` executes the same binary, cwd, environment, * and already-assembled global arguments as the imminent PTY launch. We inspect * its response only in memory and issue nothing unless every effective binding - * is exactly the recorded 0.149.1 contract. + * is exactly the recorded contract (identical for every recorded version; see + * RECORDED_PROMPT_INPUT_VERSIONS). The `01491` in the name is historical: it + * is a public export that Agent Code imports, and it now issues a profile for + * each recorded version. */ export async function prepareCodex01491PromptInputProfile( options: Codex01491PromptInputProfileOptions, @@ -83,6 +108,7 @@ export async function prepareCodex01491PromptInputProfile( return { ok: false, reason: 'effective-config-unverified' } } if (!attestation.ok) return attestation + const cliVersion = attestation.cliVersion const configOverrides = Object.freeze([ ...CODEX_01491_PROMPT_INPUT_OVERRIDES, @@ -91,8 +117,8 @@ export async function prepareCodex01491PromptInputProfile( const profile = Object.freeze({ profileVersion: 1 as const, provider: 'codex' as const, - cliVersion: '0.149.1' as const, - upstreamTag: 'rust-v0.149.1' as const, + cliVersion, + upstreamTag: RECORDED_PROMPT_INPUT_VERSIONS[cliVersion], submitKey: 'enter' as const, queueKey: 'tab' as const, vimMode: false as const, @@ -113,8 +139,8 @@ export function isIssuedCodexPromptInputProfile( const profile = value as CodexPromptInputProfile return profile.profileVersion === 1 && profile.provider === 'codex' && - profile.cliVersion === '0.149.1' && - profile.upstreamTag === 'rust-v0.149.1' && + isRecordedPromptInputVersion(profile.cliVersion) && + profile.upstreamTag === RECORDED_PROMPT_INPUT_VERSIONS[profile.cliVersion] && profile.submitKey === 'enter' && profile.queueKey === 'tab' && profile.vimMode === false && @@ -139,7 +165,7 @@ export function assertIssuedCodexPromptInputProfile( } type AttestationResult = - | { ok: true } + | { ok: true; cliVersion: RecordedPromptInputVersion } | Extract async function readEffectiveInputAttestation( @@ -159,7 +185,7 @@ async function readEffectiveInputAttestation( return await new Promise(resolve => { let settled = false let stdout = '' - let initializedVersion: string | null = null + let initializedVersion: RecordedPromptInputVersion | null = null const child = spawn(options.binary, args, { cwd: options.cwd, env: options.env, @@ -226,7 +252,7 @@ async function readEffectiveInputAttestation( ? new RegExp(`^${PROBE_CLIENT_NAME}/(\\d+\\.\\d+\\.\\d+)(?:\\s|$)`) .exec(userAgent)?.[1] ?? null : null - if (version !== '0.149.1') { + if (!isRecordedPromptInputVersion(version)) { finish({ ok: false, reason: 'unsupported-cli' }) return } @@ -245,12 +271,12 @@ async function readEffectiveInputAttestation( if (message.id !== CONFIG_READ_ID) continue if (!isSuccessfulResponse(message, CONFIG_READ_ID) || - initializedVersion !== '0.149.1' || + initializedVersion === null || !effectiveInputIsRecordedContract(message.result)) { refuse() return } - finish({ ok: true }) + finish({ ok: true, cliVersion: initializedVersion }) return } }) diff --git a/testing/fixtures/prompt-input/catalog.md b/testing/fixtures/prompt-input/catalog.md index 12e0a63..cf3892b 100644 --- a/testing/fixtures/prompt-input/catalog.md +++ b/testing/fixtures/prompt-input/catalog.md @@ -56,6 +56,34 @@ npm run build npx tsx testing/record-resume-capability-shape.mts ``` +## Codex 0.157.1 (#63) + +The same 16 cases were re-recorded against codex-cli `0.157.1` on 2026-09-27, twice: +- `codex-01571-recorded.json`: inline (`--no-alt-screen`), comparable row for row with the 0.149.1 corpus. +- `codex-01571-fullscreen-recorded.json`: fullscreen, 0.157's default and how Agent Code launches Codex. + +`codex-01571-config-read-recorded.json` and `codex-01571-config-source.json` are the matching `config/read` projection and the per-tag audit of the config precedence code. + +The recorder needed these 0.156+ adjustments. Each is gated so the 0.149.1 recording conditions are unchanged, and each is explained where it is made: +- `--no-daemon`: the isolated `CODEX_HOME` otherwise exceeds SUN_LEN for the managed daemon's socket. +- A seen `gpt-5.6-sol → gpt-6-sol` model migration. +- YAML frontmatter on the fixture skill. +- The trust case answers `1` then Enter, and asserts that `1` alone leaves the dialog up. +- All held slow-turn requests are released: 0.157 retries a request that has written no bytes. +- Resize frames are windowed on the last painted row. +- The popup wait uses the popup's own key hint. + +Provider facts that differ from 0.149.1: +- **Vim default.** A Vim-default composer now opens in Insert, so `vim-normal-default` submits `iabc`. It is outside the issued profile, which forces Vim off. +- **Skill popup.** The skill popup paints above the composer, with the hint `enter insert · esc close`. +- **Trust dialog.** The trust dialog is the new `Folder access` layout, and its last row has the idle-footer shape. + +The composer-surface classifier recognises the last two. Every issued-profile case agrees with 0.149.1. + +**Known provider flake, not hidden:** 0.157.1 sometimes drops a `?` from a single-chunk typed draft (1 in 5 runs of `ordinary-modal-sentinel-draft`; codex-headless#68). The recorder fails loudly when that happens, because rollout, request and typed text disagree. The committed corpus is from runs where the text arrived intact; the drop is tracked in #68 with its own evidence. + +**Side requests:** 0.157.1 also sends a thread-title side request after the first prompt. The fixture server answers it without holding or recording it (see `serveFixture`). + ## Inventory | Case | Provider fact recorded | @@ -76,6 +104,8 @@ npx tsx testing/record-resume-capability-shape.mts | `ordinary-vim-sentinel-cwd` | A literal `Vim: Insert` cwd suffix is ordinary footer text, not evidence that Vim mode is active. | | `lower-layer-keymap-valid-control` | The lower-layer `queue=[]` plus `toggle_shortcuts="tab"` map reaches a composer with no request or rollout user item. | | `lower-layer-keymap-issued-profile-conflict` | Adding the exact four package-issued CLI overrides makes the otherwise-valid lower map exit 1 before the composer, request, or rollout user item. | +| `slash-popup-enter-selects-command` | 0.156+ only: Enter while the slash-command popup (painted above the composer, no hint row) owns it dispatches the command and submits nothing. | +| `file-popup-enter-inserts-mention` | 0.156+ only: Enter while the file/mention popup owns it inserts the item and submits nothing. | | `capability-6244eac-recorded` | The built pre-repair package is constructible by deep import and exposes/retains raw state. | ## Source boundary diff --git a/testing/fixtures/prompt-input/codex-01491-app-server-fixture.mjs b/testing/fixtures/prompt-input/codex-01491-app-server-fixture.mjs index 4a5fc71..697f006 100644 --- a/testing/fixtures/prompt-input/codex-01491-app-server-fixture.mjs +++ b/testing/fixtures/prompt-input/codex-01491-app-server-fixture.mjs @@ -4,8 +4,11 @@ import { readFileSync } from 'node:fs' import { createInterface } from 'node:readline' import { fileURLToPath } from 'node:url' +// WHY the recording is selectable (#63): each recorded Codex version has its +// own config/read projection, and the profile tests replay each one through +// this same protocol shell. The default stays the original 0.149.1 file. const recorded = JSON.parse(readFileSync(fileURLToPath(new URL( - './codex-01491-config-read-recorded.json', + process.env.CODEX_PROFILE_FIXTURE_RECORDING ?? './codex-01491-config-read-recorded.json', import.meta.url, )), 'utf8')) const mode = process.env.CODEX_PROFILE_FIXTURE_MODE ?? 'recorded-safe' diff --git a/testing/fixtures/prompt-input/codex-01571-config-read-recorded.json b/testing/fixtures/prompt-input/codex-01571-config-read-recorded.json new file mode 100644 index 0000000..e810f43 --- /dev/null +++ b/testing/fixtures/prompt-input/codex-01571-config-read-recorded.json @@ -0,0 +1,29 @@ +{ + "schemaVersion": 1, + "provider": { + "cliVersion": "0.157.1", + "binarySha256": "27ceb5f9b957b43a519efe4eaa3816a0bffb0a531a2c89af18840c0a3c016a7d", + "upstreamTag": "rust-v0.157.1", + "upstreamCommitSha": "36650394c5b38c2990ccf2a3457165ca3e9d9726" + }, + "protocol": { + "transport": "app-server-stdio-jsonl", + "request": "config/read", + "includeLayers": true, + "clientName": "agent_code_prompt_profile_probe" + }, + "effectiveInputProjection": { + "composerSubmit": "enter", + "composerQueue": "tab", + "globalToggleVimMode": [], + "vimModeDefault": false, + "otherNonNullKeymapBindings": [], + "layerTypes": [ + "sessionFlags", + "user", + "system" + ], + "legacyManagedLayerPresent": false + }, + "sourceEvidenceSha256": "55041ea1fcc77f2240a083cb4795e26424d6ba490ada9a1f26b47eb6766f8576" +} diff --git a/testing/fixtures/prompt-input/codex-01571-config-source.json b/testing/fixtures/prompt-input/codex-01571-config-source.json new file mode 100644 index 0000000..6ea6f62 --- /dev/null +++ b/testing/fixtures/prompt-input/codex-01571-config-source.json @@ -0,0 +1,91 @@ +{ + "schemaVersion": 1, + "upstreamTag": "rust-v0.157.1", + "upstreamCommitSha": "36650394c5b38c2990ccf2a3457165ca3e9d9726", + "recordedCases": { + "validLowerLayer": "lower-layer-keymap-valid-control", + "issuedOverridesConflict": "lower-layer-keymap-issued-profile-conflict" + }, + "files": [ + { + "path": "codex-rs/config/src/config_layer_source.rs", + "sha256": "6816bf7bd44b1f2799aae30331b77a7e8231ccacdc4cd3d44d6485f9e1118364", + "coordinates": [ + { + "startLine": 30, + "endLine": 50, + "claim": "Session flags have precedence 30; legacy managed file and MDM layers have precedence 40 and 50." + } + ] + }, + { + "path": "codex-rs/config/src/loader/mod.rs", + "sha256": "0e3131c8186b391ecb425620bee2a8649d10364d062eee21a15d1cbe48dbbb0d", + "coordinates": [ + { + "startLine": 415, + "endLine": 421, + "claim": "CLI overrides are materialized as the SessionFlags layer." + }, + { + "startLine": 427, + "endLine": 470, + "claim": "Legacy managed file and MDM configuration are loaded above session flags. New in 0.157: thread config layers are inserted by their ConfigLayerSource precedence (lines 423-425, same precedence table as config_layer_source.rs); none appear in the recorded 0.157.1 config/read projection." + } + ] + }, + { + "path": "codex-rs/config/src/overrides.rs", + "sha256": "dbac7a6f68de31e74c11e9671fe0b1f68f8783c68cff5a9ba26348e0b5f86c0a", + "coordinates": [ + { + "startLine": 9, + "endLine": 14, + "claim": "Every dotted CLI override is written into one session-layer TOML value." + } + ] + }, + { + "path": "codex-rs/config/src/merge.rs", + "sha256": "aca71a56a375d83e79009b59b4c1421e70790d625694b4601cd13f182003817f", + "coordinates": [ + { + "startLine": 56, + "endLine": 58, + "claim": "The overlay passed to the TOML merge has precedence over the base." + }, + { + "startLine": 94, + "endLine": 151, + "claim": "Nested tables merge recursively and scalar/array overlays replace lower values (0.157 adds a credentials env de-duplication clause at lines 108-138 that does not touch tui keys)." + } + ] + }, + { + "path": "codex-rs/tui/src/keymap.rs", + "sha256": "006feab4bc833464cce59c11e0d09dae82b7e8f3f64e2885b3cc9e12b56e02d8", + "coordinates": [ + { + "startLine": 626, + "endLine": 636, + "claim": "Runtime keymap construction promises parse and ambiguity errors." + }, + { + "startLine": 1565, + "endLine": 1625, + "claim": "Conflict validation runs on the fully resolved effective keymap before use (resolved, configure_search, validate_conflicts, chord validation)." + }, + { + "startLine": 2005, + "endLine": 2067, + "claim": "Composer submit, queue and toggle-shortcuts participate in the same app-scope uniqueness pass (main_bindings, validate_unique(\"app\"))." + }, + { + "startLine": 2366, + "endLine": 2386, + "claim": "Duplicate effective bindings return an actionable ambiguous-keymap error." + } + ] + } + ] +} diff --git a/testing/fixtures/prompt-input/codex-01571-fullscreen-recorded.json b/testing/fixtures/prompt-input/codex-01571-fullscreen-recorded.json new file mode 100644 index 0000000..129328c --- /dev/null +++ b/testing/fixtures/prompt-input/codex-01571-fullscreen-recorded.json @@ -0,0 +1,1102 @@ +{ + "schemaVersion": 1, + "sanitizerVersion": 1, + "provider": { + "cliVersion": "codex-cli 0.157.1", + "binarySha256": "27ceb5f9b957b43a519efe4eaa3816a0bffb0a531a2c89af18840c0a3c016a7d", + "upstreamTag": "rust-v0.157.1" + }, + "terminal": { + "cols": 140, + "rows": 42 + }, + "source": { + "kind": "real-codex-tui-local-canned-responses", + "fixtureSseSha256": "66658b7a1d9b0e3b234de932f552b946b8a005520888f24e001a024fb9a29e5b" + }, + "cases": [ + { + "id": "trust-action-then-submit", + "sourceLabel": "recorded-source-05d5a85ed82f49f0", + "rawPtySha256": "05d5a85ed82f49f01d484aa095da5d636870c1785af53c84b28018072545940c", + "rolloutSha256": "f0354058eadf00b55b17ec1b179746c2a602bda133d0558b7b3e6b2faea4989b", + "rawRequestSha256": "9ccf1eaaa9572e80b4729c7c69e3276d3465e4272e7c04217de8b908cbc6a062", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "RECORDED_TRUST_PROMPT", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "RECORDED_TRUST_PROMPT", + "requestUserText": "RECORDED_TRUST_PROMPT", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "› RECORDED_TRUST_PROMPT", + "", + " GPT-5.6-Sol low · " + ], + "nonComposerWrites": [ + "1", + "\r" + ], + "modal": [ + " ", + " Folder access", + " ", + " ", + " Trust this folder? Codex can read, edit, and run files here, subject to your permission settings. Folder settings can run code", + " automatically, even without a model request. Continue only if you trust these files. Your trust decision will be saved.", + "", + "› 1. Trust and continue ", + " 2. Quit", + "", + " enter continue · esc quit" + ] + }, + { + "id": "combining-grapheme-backspace", + "sourceLabel": "recorded-source-8940f86a532a74b3", + "rawPtySha256": "8940f86a532a74b398818c2bf3392a9dd47a8923373e22a4dd376bdb7bafb431", + "rolloutSha256": "8e9ba0dde30b5a0ae7a182c74a2a00b68bcfb7774d7963367ee665d7fd0d76a7", + "rawRequestSha256": "36f5c993fcae8cd7bc242c71e2f74139db4e82ed448a3ff2a01dd1e9cc98dba4", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "A é", + "", + "X", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "A X", + "requestUserText": "A X", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "› A X", + "", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "mixed-cjk-ctrl-w", + "sourceLabel": "recorded-source-5e478fc81de02201", + "rawPtySha256": "5e478fc81de02201fcde92703387ecd6d0c983d4f0165e46f5e944198d83a132", + "rolloutSha256": "a4dcede6283e58db9535d8027e986e22bc7f7d96998440ce623840d41aad5aff", + "rawRequestSha256": "0debb2c8b5c2188537f6df8fd6aaa379ffe0110b40de3624a82f2fa88d83721d", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "abc 世界def", + "\u0017", + "X", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "abc 世界X", + "requestUserText": "abc 世界X", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "› abc 世界X", + "", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "repeated-line-boundaries", + "sourceLabel": "recorded-source-af95beb7da2bd8fa", + "rawPtySha256": "af95beb7da2bd8faa1cf1b17ca3b5a55bac7c6597755c8455a8683fa382db139", + "rolloutSha256": "9e317f2cd8a085fecea6c0776129fd26e70cb3784decfb1567ea35606a988830", + "rawRequestSha256": "cb785f3c0cf596c8f93f0b0731b81a663a05a5189bc44be799fad50054d75663", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "\u001b[200~one\ntwo\u001b[201~", + "\u0001", + "\u0001", + "X", + "\u0005", + "\u0005", + "Y", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "Xone\ntwoY", + "requestUserText": "Xone\ntwoY", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "› Xone", + " twoY", + "", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "remapped-kill-line-start", + "sourceLabel": "recorded-source-cb1044728ea6b760", + "rawPtySha256": "cb1044728ea6b7602addcfea8e9471dc627ffb71c3f9a727e2eb3632b1cf4f2e", + "rolloutSha256": "b881c167e8e02153d810eb7f0f9e23bfebea3e68e231fbb3280c68f2b6a428a4", + "rawRequestSha256": "16a1ad248f80269848d085c161bde5b78ba88cec7943932f00f4d1661257fcd2", + "configClass": "explicit-cli-override", + "lowerLayerConfig": [], + "configOverrides": [ + "tui.keymap.editor.kill_line_start=\"alt-u\"" + ], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "abc def", + "\u0015", + "X", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "abc defX", + "requestUserText": "abc defX", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "› abc defX", + "", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "vim-normal-default", + "sourceLabel": "recorded-source-0a537abfb2e6b2f7", + "rawPtySha256": "0a537abfb2e6b2f74ba53bc44d8e97b8b101ccaa12237221044ccbf1d8a47cc1", + "rolloutSha256": "538be00b39044966ba6a9983e36c190177925cf75d663ec72b9041c23a97bf4d", + "rawRequestSha256": "50dfd8190c804db4203b6ab0d4ec1fb37cb47c3154c31228f3bbb0923cb8d2c7", + "configClass": "explicit-cli-override", + "lowerLayerConfig": [], + "configOverrides": [ + "tui.vim_mode_default=true" + ], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "i", + "abc", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "iabc", + "requestUserText": "iabc", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "› iabc", + "", + " GPT-5.6-Sol low · Vim: Insert" + ] + }, + { + "id": "unbound-submit-enter", + "sourceLabel": "recorded-source-dbaac1bedf6b4e34", + "rawPtySha256": "dbaac1bedf6b4e346ad2e8b65203c8ac474edaee52963cab508e9993a96089aa", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "explicit-cli-override", + "lowerLayerConfig": [], + "configOverrides": [ + "tui.keymap.composer.submit=[]" + ], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "UNBOUND_SUBMIT", + "\r" + ], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "› UNBOUND_SUBMIT", + "", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "modal-ctrl-c-preserves-draft", + "sourceLabel": "recorded-source-b83f4484abcbb39a", + "rawPtySha256": "b83f4484abcbb39a50c8085b410d5b455e8a4270fc91947594c4e06c95087be9", + "rolloutSha256": "114c53d2fba85b387c6adf54e6ba4ea78d00b15d6cf3b8a4eaa24eddbb99526b", + "rawRequestSha256": "200c1e58614d8c581deaab5c0db4e850c6057b3b98c0315e8608baef7c5a7d05", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "recorded prefix ", + "\u0012", + "\u0003", + "suffix", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "recorded prefix suffix", + "requestUserText": "recorded prefix suffix", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "› recorded prefix suffix", + "", + " GPT-5.6-Sol low · " + ], + "popup": [ + "", + "", + "", + "", + "› recorded prefix", + "", + " GPT-5.6-Sol low · ", + " reverse-i-search: " + ] + }, + { + "id": "tab-footer-spoof-skill-popup", + "sourceLabel": "recorded-source-9a9df303391539d4", + "rawPtySha256": "9a9df303391539d4261a8bbcc2fd137e0f684dbcb438591a27688defd1b088b9", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "literal tab to queue $recorded_evi", + "\t" + ], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "screenBeforeFinalWrite": [ + "", + "no matches", + "", + " enter insert · esc close", + "", + "› literal tab to queue $recorded_evi", + "", + " GPT-5.6-Sol low · " + ], + "popup": [ + "", + "no matches", + "", + " enter insert · esc close", + "", + "› literal tab to queue $recorded_evi", + "", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "active-footer-tab-queue", + "sourceLabel": "recorded-source-0b3ac219f6cdac4e", + "rawPtySha256": "0b3ac219f6cdac4e301170b4a51b058177349879c9d038ad90f2e04cca2b8356", + "rolloutSha256": "19d2b19d8c67e40084d2d3d0b8973a4253e2a8d6826d6e976d36bcc48d89ead4", + "rawRequestSha256": "ea03a4f90a0df973bae97da8bbba1c8e33d79c192f04bbf10f1d2395f0909df1", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "RECORDED_QUEUED_PROMPT", + "\t" + ], + "expectedSubmission": true, + "durableUserText": "RECORDED_QUEUED_PROMPT", + "requestUserText": "RECORDED_QUEUED_PROMPT", + "screenBeforeFinalWrite": [ + "", + " Working ( • esc to interrupt)", + "", + "", + "› RECORDED_QUEUED_PROMPT", + "", + " GPT-5.6-Sol low · ", + " tab to queue message" + ], + "activeTurnFooter": [ + "", + " Working ( • esc to interrupt)", + "", + "", + "› Ask Codex to do anything", + "", + " GPT-5.6-Sol low · · ⠇", + " ? for shortcuts" + ] + }, + { + "id": "narrow-soft-wrap-resize-redraw", + "sourceLabel": "recorded-source-357b0d3f57bf3138", + "rawPtySha256": "357b0d3f57bf3138f87dc81f9763ea680ec4e86ce473b666e6f33088c86ff98d", + "rolloutSha256": "5ec216aa5a253f01eb89ac2444f473c7b31001f151577cd10ee2bf4e601986f7", + "rawRequestSha256": "cd561f0fee68e05e7b125ed401ce463787d4aaf929d4f1dfdb28cd10df675913", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 52, + "rows": 24 + }, + "inputChunks": [ + "RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + "requestUserText": "RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "› RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + "", + " GPT-5.6-Sol low · " + ], + "beforeTypingFrame": { + "generation": 30, + "cols": 52, + "cursor": { + "x": 2, + "y": 20 + }, + "rows": [ + { + "viewportRow": 14, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 15, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 16, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 17, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 18, + "text": " Tip: Press f3 to search this conversation.", + "isWrapped": false + }, + { + "viewportRow": 19, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 20, + "text": "› Ask Codex to do anything", + "isWrapped": false + }, + { + "viewportRow": 21, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 22, + "text": " GPT-5.6-Sol low · ", + "isWrapped": false + }, + { + "viewportRow": 23, + "text": " ? for shortcuts", + "isWrapped": false + } + ] + }, + "resizeTrace": { + "requested": { + "cols": 92, + "rows": 24 + }, + "rawChunkCountBeforeResize": 59, + "rawChunkCountBeforeProviderRedraw": 59, + "rawChunkCountAfterProviderRedraw": 63, + "preRedrawGenerationUnchanged": true, + "postRedrawGenerationAdvanced": true, + "narrow": { + "generation": 59, + "cols": 52, + "cursor": { + "x": 41, + "y": 20 + }, + "rows": [ + { + "viewportRow": 13, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 14, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 15, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 16, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 17, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 18, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 19, + "text": "› RECORDED_NARROW_WRAP alpha beta gamma delta", + "isWrapped": false + }, + { + "viewportRow": 20, + "text": " epsilon zeta eta theta iota kappa omega", + "isWrapped": false + }, + { + "viewportRow": 21, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 22, + "text": " GPT-5.6-Sol low · ", + "isWrapped": false + } + ] + }, + "beforeProviderRedraw": { + "generation": 59, + "cols": 92, + "cursor": { + "x": 41, + "y": 20 + }, + "rows": [ + { + "viewportRow": 13, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 14, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 15, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 16, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 17, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 18, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 19, + "text": "› RECORDED_NARROW_WRAP alpha beta gamma delta", + "isWrapped": false + }, + { + "viewportRow": 20, + "text": " epsilon zeta eta theta iota kappa omega", + "isWrapped": false + }, + { + "viewportRow": 21, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 22, + "text": " GPT-5.6-Sol low · ", + "isWrapped": false + } + ] + }, + "afterProviderRedraw": { + "generation": 63, + "cols": 92, + "cursor": { + "x": 85, + "y": 20 + }, + "rows": [ + { + "viewportRow": 13, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 14, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 15, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 16, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 17, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 18, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 19, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 20, + "text": "› RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + "isWrapped": false + }, + { + "viewportRow": 21, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 22, + "text": " GPT-5.6-Sol low · ", + "isWrapped": false + } + ] + } + } + }, + { + "id": "unchanged-redraw-after-edit", + "sourceLabel": "recorded-source-193f0c963269d1dc", + "rawPtySha256": "193f0c963269d1dc8cda4a85dc1dd7e401172bda2d08c90a043b594ce2db3887", + "rolloutSha256": "5a1293ecbac4bdce1f73b5dc66405ae7080bef8aefc9536b9ffc58afb52b0798", + "rawRequestSha256": "2f2694c1f481a85dfc6828f7cb9ed14e8fea5d18284dff44c8cb650a6030e19a", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "\r" + ], + "expectedSubmission": true, + "durableUserText": "RECORDED_UNCHANGED_BASE_EDIT", + "requestUserText": "RECORDED_UNCHANGED_BASE_EDIT", + "screenBeforeFinalWrite": [ + "", + "", + " ", + "", + "", + "› RECORDED_UNCHANGED_BASE_EDIT", + "", + " GPT-5.6-Sol low · " + ], + "editAcknowledgementTrace": { + "baseDraft": "RECORDED_UNCHANGED_BASE", + "edit": "_EDIT", + "setupDurableUserText": "RECORDED_SLOW_TURN", + "setupRequestUserText": "RECORDED_SLOW_TURN", + "rawChunkCountBeforeEdit": 60, + "rawChunkCountAtUnchangedRedraw": 61, + "schedulingControl": "SIGSTOP_AFTER_EDIT_WRITE_BEFORE_PROVIDER_PARSE", + "beforeEdit": { + "generation": 60, + "cols": 140, + "cursor": { + "x": 25, + "y": 38 + }, + "rows": [ + { + "viewportRow": 32, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 33, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 34, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 35, + "text": " Working ( • esc to interrupt)", + "isWrapped": false + }, + { + "viewportRow": 36, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 37, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 38, + "text": "› RECORDED_UNCHANGED_BASE", + "isWrapped": false + }, + { + "viewportRow": 39, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 40, + "text": " GPT-5.6-Sol low · · ⠇", + "isWrapped": false + }, + { + "viewportRow": 41, + "text": " tab to queue message", + "isWrapped": false + } + ] + }, + "unchangedAfterEdit": { + "generation": 61, + "cols": 140, + "cursor": { + "x": 25, + "y": 38 + }, + "rows": [ + { + "viewportRow": 32, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 33, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 34, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 35, + "text": " Working ( • esc to interrupt)", + "isWrapped": false + }, + { + "viewportRow": 36, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 37, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 38, + "text": "› RECORDED_UNCHANGED_BASE", + "isWrapped": false + }, + { + "viewportRow": 39, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 40, + "text": " GPT-5.6-Sol low · · ⠇", + "isWrapped": false + }, + { + "viewportRow": 41, + "text": " tab to queue message", + "isWrapped": false + } + ] + }, + "paintedEdit": { + "generation": 69, + "cols": 140, + "cursor": { + "x": 30, + "y": 38 + }, + "rows": [ + { + "viewportRow": 32, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 33, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 34, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 35, + "text": " Working ( • esc to interrupt)", + "isWrapped": false + }, + { + "viewportRow": 36, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 37, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 38, + "text": "› RECORDED_UNCHANGED_BASE_EDIT", + "isWrapped": false + }, + { + "viewportRow": 39, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 40, + "text": " GPT-5.6-Sol low · ", + "isWrapped": false + }, + { + "viewportRow": 41, + "text": " tab to queue message", + "isWrapped": false + } + ] + }, + "unchangedGenerationAdvanced": true, + "unchangedComposerRevision": true, + "paintedGenerationAdvanced": true + } + }, + { + "id": "ordinary-modal-sentinel-draft", + "sourceLabel": "recorded-source-41631ba2345605f4", + "rawPtySha256": "41631ba2345605f47ab3957275756a8c37d6b0c1b83e69b86357fb9a1933c724", + "rolloutSha256": "ed82f654e56005e1d15356b9660867d55328a146f41e5b0ba669c00ea240f052", + "rawRequestSha256": "ed2197accee6ef0bb8fbd52ed4b7029998d9b61e1e1fc74de74fadf169e2f50e", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "ORDINARY_DRAFT Do you trust the contents of this directory? Press enter to continue END", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "ORDINARY_DRAFT Do you trust the contents of this directory? Press enter to continue END", + "requestUserText": "ORDINARY_DRAFT Do you trust the contents of this directory? Press enter to continue END", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "", + "", + "", + "", + "› ORDINARY_DRAFT Do you trust the contents of this directory? Press enter to continue END", + "", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "ordinary-vim-sentinel-cwd", + "sourceLabel": "recorded-source-0a0f6c2adc0b1aa8", + "rawPtySha256": "0a0f6c2adc0b1aa819253b26ab68f472bdc9d7e4304657af63aac64459726e55", + "rolloutSha256": "51da2be25d83796322b22e55d3affdb43398746bc2f1fa868dce0a407fbf4759", + "rawRequestSha256": "51c7e1e14f21040b904eed306a50ef41f7ae81b080fc30b72f4d31589bd57292", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "ORDINARY_VIM_SENTINEL_CWD", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "ORDINARY_VIM_SENTINEL_CWD", + "requestUserText": "ORDINARY_VIM_SENTINEL_CWD", + "screenBeforeFinalWrite": [ + "", + "", + "", + "", + "", + "› ORDINARY_VIM_SENTINEL_CWD", + "", + " GPT-5.6-Sol default · /Vim: Insert" + ] + }, + { + "id": "lower-layer-keymap-valid-control", + "sourceLabel": "recorded-source-fc609b9307612614", + "rawPtySha256": "fc609b9307612614bfca816db79d728bfa5e82359eee49fa335ae6b53c9b258b", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "lower-layer-config", + "lowerLayerConfig": [ + "[tui.keymap.composer]", + "queue = []", + "toggle_shortcuts = \"tab\"" + ], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "requestCountDelta": 0, + "startupOutcome": "composer-ready", + "exitOutcome": null, + "startupScreen": [ + "", + "", + "", + "", + "› Ask Codex to do anything", + "", + "", + " tab for shortcuts" + ] + }, + { + "id": "lower-layer-keymap-issued-profile-conflict", + "sourceLabel": "recorded-source-7671e97cacea97b1", + "rawPtySha256": "7671e97cacea97b17c5d26a33e4f3492dcb8147b924cc333df00b880efe5fe59", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "lower-layer-plus-issued-cli-override", + "lowerLayerConfig": [ + "[tui.keymap.composer]", + "queue = []", + "toggle_shortcuts = \"tab\"" + ], + "configOverrides": [ + "tui.keymap.composer.submit=\"enter\"", + "tui.keymap.composer.queue=\"tab\"", + "tui.vim_mode_default=false", + "tui.keymap.global.toggle_vim_mode=[]" + ], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "requestCountDelta": 0, + "startupOutcome": "rejected-before-composer", + "exitOutcome": { + "exitCode": 1, + "signal": 0 + }, + "startupScreen": [ + "Error: Ambiguous `tui.keymap.app` bindings: `composer.queue` and `composer.toggle_shortcuts` use the same key. Set unique keys in `~/.codex/", + "config.toml` and retry. See the Codex keymap documentation for supported actions and examples." + ] + }, + { + "id": "slash-popup-enter-selects-command", + "sourceLabel": "recorded-source-b8630e6656578466", + "rawPtySha256": "b8630e6656578466c248584fca14e9eb2cde9f8f88e3f37a4f682d30f5e9080d", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "/stat", + "\r" + ], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "screenBeforeFinalWrite": [ + "", + "", + "› /status show current session configuration and token usage ", + " /statusline configure which items appear in the status line", + "", + "› /stat", + "", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "file-popup-enter-inserts-mention", + "sourceLabel": "recorded-source-dd0f4a9d621939bb", + "rawPtySha256": "dd0f4a9d621939bb544474eb08d41c8ffd9a1b5ed10dadaad3d8d126b5780605", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "@READ", + "\r" + ], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "screenBeforeFinalWrite": [ + "", + "", + "", + " enter/tab insert · esc close · ↑/↓ select · ←/→ filter", + "", + "› @READ", + "", + " GPT-5.6-Sol low · " + ] + } + ] +} diff --git a/testing/fixtures/prompt-input/codex-01571-recorded.json b/testing/fixtures/prompt-input/codex-01571-recorded.json new file mode 100644 index 0000000..971a50f --- /dev/null +++ b/testing/fixtures/prompt-input/codex-01571-recorded.json @@ -0,0 +1,1102 @@ +{ + "schemaVersion": 1, + "sanitizerVersion": 1, + "provider": { + "cliVersion": "codex-cli 0.157.1", + "binarySha256": "27ceb5f9b957b43a519efe4eaa3816a0bffb0a531a2c89af18840c0a3c016a7d", + "upstreamTag": "rust-v0.157.1" + }, + "terminal": { + "cols": 140, + "rows": 42 + }, + "source": { + "kind": "real-codex-tui-local-canned-responses", + "fixtureSseSha256": "66658b7a1d9b0e3b234de932f552b946b8a005520888f24e001a024fb9a29e5b" + }, + "cases": [ + { + "id": "trust-action-then-submit", + "sourceLabel": "recorded-source-bb6fcdefc3edc4e6", + "rawPtySha256": "bb6fcdefc3edc4e636b6082e070903744dbace4491b8253bccc4632b762824d6", + "rolloutSha256": "bd63699b0d2dfd66ca6c46ac3c863d0557af1e9be5fc69b308a996787e930589", + "rawRequestSha256": "ce82e2129feec9d52e26ba5939386a3b2dfefdc5a4fd07b6bec79a11ce8ccd72", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "RECORDED_TRUST_PROMPT", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "RECORDED_TRUST_PROMPT", + "requestUserText": "RECORDED_TRUST_PROMPT", + "screenBeforeFinalWrite": [ + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› RECORDED_TRUST_PROMPT", + " ", + " GPT-5.6-Sol low · " + ], + "nonComposerWrites": [ + "1", + "\r" + ], + "modal": [ + "", + " Folder access", + " ", + "", + " Trust this folder? Codex can read, edit, and run files here, subject to your permission settings. Folder settings can run code", + " automatically, even without a model request. Continue only if you trust these files. Your trust decision will be saved.", + "", + "› 1. Trust and continue ", + " 2. Quit", + "", + " enter continue · esc quit" + ] + }, + { + "id": "combining-grapheme-backspace", + "sourceLabel": "recorded-source-57671a2aa17a9638", + "rawPtySha256": "57671a2aa17a9638da28a9bc84dd93e38a6127f19d5f36bf82e32fee29990555", + "rolloutSha256": "3756c060bf4b56489fb70438c4ae5cf5ad9673e5bd6f8671a050d5fcc6df910a", + "rawRequestSha256": "80d28b79c2c2ebd61f762ea52adbdf95771d2a7de24cf4042f8af25f64287a6d", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "A é", + "", + "X", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "A X", + "requestUserText": "A X", + "screenBeforeFinalWrite": [ + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› A X", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "mixed-cjk-ctrl-w", + "sourceLabel": "recorded-source-e32aecee00f41bad", + "rawPtySha256": "e32aecee00f41bad3a9b9299891b3c966d554c0436f131735db16425cce19cd7", + "rolloutSha256": "838eb2bc211e5f5c9a17144935eab64ca95d06a6d17c90d5a634fff02b15b749", + "rawRequestSha256": "1d21d268b6ac4243bcd778f7a865b7d3ffc5521a3219eac8ff9df95015bfa1c0", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "abc 世界def", + "\u0017", + "X", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "abc 世界X", + "requestUserText": "abc 世界X", + "screenBeforeFinalWrite": [ + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› abc 世界X", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "repeated-line-boundaries", + "sourceLabel": "recorded-source-0da2d21741620169", + "rawPtySha256": "0da2d21741620169d57b13a45af7bcf7866bdff2e74bc7d0c088b981fb2b4943", + "rolloutSha256": "46651882ebbffc9c422d9f3b5b5b5852c57a5920eada29f80bfe90820b6897eb", + "rawRequestSha256": "a7d87d3fe5caea3c2b4c774cff5b7dc657b7a5c0eb527f991ae103bd28b41605", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "\u001b[200~one\ntwo\u001b[201~", + "\u0001", + "\u0001", + "X", + "\u0005", + "\u0005", + "Y", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "Xone\ntwoY", + "requestUserText": "Xone\ntwoY", + "screenBeforeFinalWrite": [ + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› Xone", + " twoY", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "remapped-kill-line-start", + "sourceLabel": "recorded-source-929cc1f4f95842ea", + "rawPtySha256": "929cc1f4f95842ea1eea1797aaa8b46110aad80c121d14da4b662eb8af2315ad", + "rolloutSha256": "24ff98b8ea08c903e580d3dee98477dcf57e2b4b8081fde24bfbc76f8ce5dd68", + "rawRequestSha256": "cd504d26c0d82c2fcd743d4fa065ecdd48517b8bafd2391ab91fcdab33d25a21", + "configClass": "explicit-cli-override", + "lowerLayerConfig": [], + "configOverrides": [ + "tui.keymap.editor.kill_line_start=\"alt-u\"" + ], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "abc def", + "\u0015", + "X", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "abc defX", + "requestUserText": "abc defX", + "screenBeforeFinalWrite": [ + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› abc defX", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "vim-normal-default", + "sourceLabel": "recorded-source-c3cbd1a723c869d0", + "rawPtySha256": "c3cbd1a723c869d096c22c9792cec2d2d497df39aae83d89efd930dd83ecaf49", + "rolloutSha256": "34a1cff8942fdf1caa1485a0f917aed8a8a5e95f6ec8e1ead1a5a91f3bf50140", + "rawRequestSha256": "b587dccede9b7932a524f97116cfc47b1b0f8cb73d48291743563bc4eb348eff", + "configClass": "explicit-cli-override", + "lowerLayerConfig": [], + "configOverrides": [ + "tui.vim_mode_default=true" + ], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "i", + "abc", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "iabc", + "requestUserText": "iabc", + "screenBeforeFinalWrite": [ + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› iabc", + " ", + " GPT-5.6-Sol low · Vim: Insert" + ] + }, + { + "id": "unbound-submit-enter", + "sourceLabel": "recorded-source-8c539d37df20c818", + "rawPtySha256": "8c539d37df20c8185fbdd79f234d8b22a498b349fa68e3392926a38bd29413a2", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "explicit-cli-override", + "lowerLayerConfig": [], + "configOverrides": [ + "tui.keymap.composer.submit=[]" + ], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "UNBOUND_SUBMIT", + "\r" + ], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "screenBeforeFinalWrite": [ + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: Use /hooks to view and manage lifecycle hooks.", + " ", + " ", + "› UNBOUND_SUBMIT", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "modal-ctrl-c-preserves-draft", + "sourceLabel": "recorded-source-0b6766e35802cf98", + "rawPtySha256": "0b6766e35802cf9805ce223db7856c5ddf3cc1be18fd3d0744c63f08dce6fff6", + "rolloutSha256": "b4fca4dba1d766971f56761ef661bf22ef1acbb6d333087fb0f49c07262940ae", + "rawRequestSha256": "eaf4896051e4152b43e160b5263a1db2330b64d6b676622502b130814015d3ae", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "recorded prefix ", + "\u0012", + "\u0003", + "suffix", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "recorded prefix suffix", + "requestUserText": "recorded prefix suffix", + "screenBeforeFinalWrite": [ + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› recorded prefix suffix", + " ", + " GPT-5.6-Sol low · " + ], + "popup": [ + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› recorded prefix", + " ", + " reverse-i-search: " + ] + }, + { + "id": "tab-footer-spoof-skill-popup", + "sourceLabel": "recorded-source-60e4b6a07edddcb2", + "rawPtySha256": "60e4b6a07edddcb2d52f4e536b1d5d88b7f6deeea5ed4c2adb6e8c5ddc385691", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "literal tab to queue $recorded_evi", + "\t" + ], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "screenBeforeFinalWrite": [ + " ", + "no matches", + " ", + " enter insert · esc close", + " ", + "› literal tab to queue $recorded_evi", + " ", + " GPT-5.6-Sol low · " + ], + "popup": [ + " ", + "no matches", + " ", + " enter insert · esc close", + " ", + "› literal tab to queue $recorded_evi", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "active-footer-tab-queue", + "sourceLabel": "recorded-source-31af2a8155d082a8", + "rawPtySha256": "31af2a8155d082a849a54f3bc9db001ebc8cee98f2f047e2165423640141cb84", + "rolloutSha256": "08aada9e8a184ebf7fcca9175bedfc3769a054cd1446b1c7ccb2662b2bb908e8", + "rawRequestSha256": "95feeff62c62b03acba64e4f7bfa4d9043419581ce3203798c20cf87263732a4", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "RECORDED_QUEUED_PROMPT", + "\t" + ], + "expectedSubmission": true, + "durableUserText": "RECORDED_QUEUED_PROMPT", + "requestUserText": "RECORDED_QUEUED_PROMPT", + "screenBeforeFinalWrite": [ + "", + " ", + " Working ( • esc to interrupt)", + " ", + " ", + "› RECORDED_QUEUED_PROMPT", + " ", + " tab to queue message 100% context left" + ], + "activeTurnFooter": [ + "", + " ", + " Working ( • esc to interrupt)", + " ", + " ", + "› Ask Codex to do anything", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "narrow-soft-wrap-resize-redraw", + "sourceLabel": "recorded-source-fa5da550e6e40faf", + "rawPtySha256": "fa5da550e6e40faf04144e1c4e2bd0ddc3825220b3196ace083a1edbb200f577", + "rolloutSha256": "708e13c9b804e79cb9c17487f61e5ba1f21b777e064d369bb0367c4b537408fb", + "rawRequestSha256": "644c05a6e1abecd6f7b50a70b77dbed62d60c492353de8ee8a79479c2cf32d26", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 52, + "rows": 24 + }, + "inputChunks": [ + "RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + "requestUserText": "RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + "screenBeforeFinalWrite": [ + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit", + " https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + " ", + " GPT-5.6-Sol low · " + ], + "beforeTypingFrame": { + "generation": 32, + "cols": 52, + "cursor": { + "x": 2, + "y": 12 + }, + "rows": [ + { + "viewportRow": 5, + "text": "╰──────────────────────────────────────────────────╯", + "isWrapped": false + }, + { + "viewportRow": 6, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 7, + "text": " Tip: New Build faster with the Desktop app. Run", + "isWrapped": false + }, + { + "viewportRow": 8, + "text": " 'codex app' or visit", + "isWrapped": false + }, + { + "viewportRow": 9, + "text": " https://chatgpt.com/codex?app-landing-page=true", + "isWrapped": false + }, + { + "viewportRow": 10, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 11, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 12, + "text": "› Ask Codex to do anything", + "isWrapped": false + }, + { + "viewportRow": 13, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 14, + "text": " GPT-5.6-Sol low · ", + "isWrapped": false + } + ] + }, + "resizeTrace": { + "requested": { + "cols": 92, + "rows": 24 + }, + "rawChunkCountBeforeResize": 33, + "rawChunkCountBeforeProviderRedraw": 33, + "rawChunkCountAfterProviderRedraw": 38, + "preRedrawGenerationUnchanged": true, + "postRedrawGenerationAdvanced": true, + "narrow": { + "generation": 33, + "cols": 52, + "cursor": { + "x": 41, + "y": 13 + }, + "rows": [ + { + "viewportRow": 6, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 7, + "text": " Tip: New Build faster with the Desktop app. Run", + "isWrapped": false + }, + { + "viewportRow": 8, + "text": " 'codex app' or visit", + "isWrapped": false + }, + { + "viewportRow": 9, + "text": " https://chatgpt.com/codex?app-landing-page=true", + "isWrapped": false + }, + { + "viewportRow": 10, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 11, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 12, + "text": "› RECORDED_NARROW_WRAP alpha beta gamma delta", + "isWrapped": false + }, + { + "viewportRow": 13, + "text": " epsilon zeta eta theta iota kappa omega", + "isWrapped": false + }, + { + "viewportRow": 14, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 15, + "text": " GPT-5.6-Sol low · ", + "isWrapped": false + } + ] + }, + "beforeProviderRedraw": { + "generation": 33, + "cols": 92, + "cursor": { + "x": 41, + "y": 13 + }, + "rows": [ + { + "viewportRow": 6, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 7, + "text": " Tip: New Build faster with the Desktop app. Run", + "isWrapped": false + }, + { + "viewportRow": 8, + "text": " 'codex app' or visit", + "isWrapped": false + }, + { + "viewportRow": 9, + "text": " https://chatgpt.com/codex?app-landing-page=true", + "isWrapped": false + }, + { + "viewportRow": 10, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 11, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 12, + "text": "› RECORDED_NARROW_WRAP alpha beta gamma delta", + "isWrapped": false + }, + { + "viewportRow": 13, + "text": " epsilon zeta eta theta iota kappa omega", + "isWrapped": false + }, + { + "viewportRow": 14, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 15, + "text": " GPT-5.6-Sol low · ", + "isWrapped": false + } + ] + }, + "afterProviderRedraw": { + "generation": 38, + "cols": 92, + "cursor": { + "x": 85, + "y": 11 + }, + "rows": [ + { + "viewportRow": 4, + "text": "│ directory: │", + "isWrapped": false + }, + { + "viewportRow": 5, + "text": "╰──────────────────────────────────────────────────────────╯", + "isWrapped": false + }, + { + "viewportRow": 6, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 7, + "text": " Tip: New Build faster with the Desktop app. Run 'codex app' or visit", + "isWrapped": false + }, + { + "viewportRow": 8, + "text": " https://chatgpt.com/codex?app-landing-page=true", + "isWrapped": false + }, + { + "viewportRow": 9, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 10, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 11, + "text": "› RECORDED_NARROW_WRAP alpha beta gamma delta epsilon zeta eta theta iota kappa omega", + "isWrapped": false + }, + { + "viewportRow": 12, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 13, + "text": " GPT-5.6-Sol low · ", + "isWrapped": false + } + ] + } + } + }, + { + "id": "unchanged-redraw-after-edit", + "sourceLabel": "recorded-source-eb8b434b3854c8cd", + "rawPtySha256": "eb8b434b3854c8cd1acb7064f1922c70a62f1458abf9ab9953e2dffda9b562c6", + "rolloutSha256": "9c057ab816c22dbf1576aee52cbbfa1ff5793536ea123a38adf221c39158e974", + "rawRequestSha256": "2d1d8e08304f9fc0f28c8ca3b93bb19c4a6769094f53ca21fee16fda780e18c3", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "\r" + ], + "expectedSubmission": true, + "durableUserText": "RECORDED_UNCHANGED_BASE_EDIT", + "requestUserText": "RECORDED_UNCHANGED_BASE_EDIT", + "screenBeforeFinalWrite": [ + "", + "", + " 01:42", + " ", + " ", + "› RECORDED_UNCHANGED_BASE_EDIT", + " ", + " GPT-5.6-Sol low · " + ], + "editAcknowledgementTrace": { + "baseDraft": "RECORDED_UNCHANGED_BASE", + "edit": "_EDIT", + "setupDurableUserText": "RECORDED_SLOW_TURN", + "setupRequestUserText": "RECORDED_SLOW_TURN", + "rawChunkCountBeforeEdit": 62, + "rawChunkCountAtUnchangedRedraw": 63, + "schedulingControl": "SIGSTOP_AFTER_EDIT_WRITE_BEFORE_PROVIDER_PARSE", + "beforeEdit": { + "generation": 62, + "cols": 140, + "cursor": { + "x": 25, + "y": 16 + }, + "rows": [ + { + "viewportRow": 9, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 10, + "text": "› RECORDED_SLOW_TURN", + "isWrapped": false + }, + { + "viewportRow": 11, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 12, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 13, + "text": " Working ( • esc to interrupt)", + "isWrapped": false + }, + { + "viewportRow": 14, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 15, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 16, + "text": "› RECORDED_UNCHANGED_BASE", + "isWrapped": false + }, + { + "viewportRow": 17, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 18, + "text": " tab to queue message 100% context left", + "isWrapped": false + } + ] + }, + "unchangedAfterEdit": { + "generation": 63, + "cols": 140, + "cursor": { + "x": 25, + "y": 16 + }, + "rows": [ + { + "viewportRow": 9, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 10, + "text": "› RECORDED_SLOW_TURN", + "isWrapped": false + }, + { + "viewportRow": 11, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 12, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 13, + "text": " Working ( • esc to interrupt)", + "isWrapped": false + }, + { + "viewportRow": 14, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 15, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 16, + "text": "› RECORDED_UNCHANGED_BASE", + "isWrapped": false + }, + { + "viewportRow": 17, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 18, + "text": " tab to queue message 100% context left", + "isWrapped": false + } + ] + }, + "paintedEdit": { + "generation": 66, + "cols": 140, + "cursor": { + "x": 30, + "y": 16 + }, + "rows": [ + { + "viewportRow": 9, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 10, + "text": "› RECORDED_SLOW_TURN", + "isWrapped": false + }, + { + "viewportRow": 11, + "text": "", + "isWrapped": false + }, + { + "viewportRow": 12, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 13, + "text": " Working ( • esc to interrupt)", + "isWrapped": false + }, + { + "viewportRow": 14, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 15, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 16, + "text": "› RECORDED_UNCHANGED_BASE_EDIT", + "isWrapped": false + }, + { + "viewportRow": 17, + "text": " ", + "isWrapped": false + }, + { + "viewportRow": 18, + "text": " tab to queue message 100% context left", + "isWrapped": false + } + ] + }, + "unchangedGenerationAdvanced": true, + "unchangedComposerRevision": true, + "paintedGenerationAdvanced": true + } + }, + { + "id": "ordinary-modal-sentinel-draft", + "sourceLabel": "recorded-source-d2806a249c852d29", + "rawPtySha256": "d2806a249c852d29b0bb0d4a5959b568b7e9ea5e8b38d437ac1f1b7cc0116c53", + "rolloutSha256": "0ba9cab7da82bb3aed33c23f43fd3fbd0c439021715f2e862f384f75771f9b09", + "rawRequestSha256": "f1fe707287a3c067e0a6f2597e4664d59ba8db6ed873eed8bd45754fcb9ccf82", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "ORDINARY_DRAFT Do you trust the contents of this directory? Press enter to continue END", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "ORDINARY_DRAFT Do you trust the contents of this directory? Press enter to continue END", + "requestUserText": "ORDINARY_DRAFT Do you trust the contents of this directory? Press enter to continue END", + "screenBeforeFinalWrite": [ + "│ >_ OpenAI Codex (v0.157.1) │", + "│ │", + "│ model: GPT-5.6-Sol low /model to change │", + "│ directory: │", + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› ORDINARY_DRAFT Do you trust the contents of this directory? Press enter to continue END", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "ordinary-vim-sentinel-cwd", + "sourceLabel": "recorded-source-c8143ea8bbffa37a", + "rawPtySha256": "c8143ea8bbffa37ad98e3e682eb848d9f8bf2b4335e92abdc740c4fc230dc90e", + "rolloutSha256": "e7b4fdab9667b9bd92b9329b3eca040ff88555ad222ebd73600c443c62f07283", + "rawRequestSha256": "1c6391ac1a03f12bb6588d9c62e8a05afd514da8867fb4740f8d11fc41658f69", + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "ORDINARY_VIM_SENTINEL_CWD", + "\r" + ], + "expectedSubmission": true, + "durableUserText": "ORDINARY_VIM_SENTINEL_CWD", + "requestUserText": "ORDINARY_VIM_SENTINEL_CWD", + "screenBeforeFinalWrite": [ + "╰──────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› ORDINARY_VIM_SENTINEL_CWD", + " ", + " GPT-5.6-Sol low · /Vim: Insert" + ] + }, + { + "id": "lower-layer-keymap-valid-control", + "sourceLabel": "recorded-source-f8fb8b7d929ddc0f", + "rawPtySha256": "f8fb8b7d929ddc0f2efb828426b803b78822e0a2cb40b48239f895df4f4ba9a8", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "lower-layer-config", + "lowerLayerConfig": [ + "[tui.keymap.composer]", + "queue = []", + "toggle_shortcuts = \"tab\"" + ], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "requestCountDelta": 0, + "startupOutcome": "composer-ready", + "exitOutcome": null, + "startupScreen": [ + "╰──────────────────────────────────────────────────────────╯", + "", + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + " ", + "› Ask Codex to do anything", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "lower-layer-keymap-issued-profile-conflict", + "sourceLabel": "recorded-source-7671e97cacea97b1", + "rawPtySha256": "7671e97cacea97b17c5d26a33e4f3492dcb8147b924cc333df00b880efe5fe59", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "lower-layer-plus-issued-cli-override", + "lowerLayerConfig": [ + "[tui.keymap.composer]", + "queue = []", + "toggle_shortcuts = \"tab\"" + ], + "configOverrides": [ + "tui.keymap.composer.submit=\"enter\"", + "tui.keymap.composer.queue=\"tab\"", + "tui.vim_mode_default=false", + "tui.keymap.global.toggle_vim_mode=[]" + ], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "requestCountDelta": 0, + "startupOutcome": "rejected-before-composer", + "exitOutcome": { + "exitCode": 1, + "signal": 0 + }, + "startupScreen": [ + "Error: Ambiguous `tui.keymap.app` bindings: `composer.queue` and `composer.toggle_shortcuts` use the same key. Set unique keys in `~/.codex/", + "config.toml` and retry. See the Codex keymap documentation for supported actions and examples." + ] + }, + { + "id": "slash-popup-enter-selects-command", + "sourceLabel": "recorded-source-30f3183257dca4bc", + "rawPtySha256": "30f3183257dca4bc58607ba9833048e542f5bea18dcebff6eebfd4bb17ac8cb6", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "/stat", + "\r" + ], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "screenBeforeFinalWrite": [ + " Tip: New Build faster with the Desktop app. Run 'codex app' or visit https://chatgpt.com/codex?app-landing-page=true", + " ", + "› /status show current session configuration and token usage ", + " /statusline configure which items appear in the status line", + " ", + "› /stat", + " ", + " GPT-5.6-Sol low · " + ] + }, + { + "id": "file-popup-enter-inserts-mention", + "sourceLabel": "recorded-source-7c017f846c733ea9", + "rawPtySha256": "7c017f846c733ea9390ccb673e50cd78a0b7b121d5e76581f5cf5c9a463d5861", + "rolloutSha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "rawRequestSha256": null, + "configClass": "recorded-default-01491", + "lowerLayerConfig": [], + "configOverrides": [], + "terminal": { + "cols": 140, + "rows": 42 + }, + "inputChunks": [ + "@READ", + "\r" + ], + "expectedSubmission": false, + "durableUserText": null, + "requestUserText": null, + "screenBeforeFinalWrite": [ + " ", + " ", + " ", + " enter/tab insert · esc close · ↑/↓ select · ←/→ filter", + " ", + "› @READ", + " ", + " GPT-5.6-Sol low · " + ] + } + ] +} diff --git a/testing/fixtures/trust-dialog-0157/folder-access-back.json b/testing/fixtures/trust-dialog-0157/folder-access-back.json new file mode 100644 index 0000000..e387d59 --- /dev/null +++ b/testing/fixtures/trust-dialog-0157/folder-access-back.json @@ -0,0 +1,163 @@ +{ +"source": "codex-cli 0.157.1 (standalone ~/.local/bin/codex), spawned through node-pty at its default 80x24 in a fresh untrusted scratch directory on 2026-09-27. No keystroke was sent: the dialog was observed, never answered, and the process was killed after 9 s (the observed marker). Redacted at equal length: the account name, the session uuid and the directory suffix inside the scratch path.", +"cols": 80, +"rows": 24, +"events": [ +{ +"t": 52, +"dir": "out", +"data": "\u001b[?2004h\u001b[>4;0m\u001b[>7u\u001b[?1004h" +}, +{ +"t": 53, +"dir": "out", +"data": "\u001b[6n\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[?u\u001b[c" +}, +{ +"t": 302, +"dir": "out", +"data": "\u001b[?2026h\u001b[?25l\u001b[?1049h\u001b[>4;0m\u001b[>7u\u001b[?1007l\u001b[?1000h\u001b[?1002h\u001b[?1006h\u001b[?1003h\u001b[?25l" +}, +{ +"t": 302, +"dir": "out", +"data": "\u001b[1;1H\u001b[J" +}, +{ +"t": 303, +"dir": "out", +"data": "\u001b[?2026h\u001b[?25l\u001b[1;1H\u001b[2m\u256d\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u256e\u001b[2;1H\u2502 >_ \u001b[22m\u001b[1mOpenAI Codex\u001b[22m\u001b[2m\u001b[2m (v0.157.1) \u2502\u001b[3;1H\u2502 \u2502\u001b[4;1H\u2502 model: \u001b[3mloading\u001b[23m \u001b[22m\u001b[38;2;99;168;248;49m/model\u001b[2m\u001b[39;49m to change \u2502\u001b[5;1H\u2502 directory: \u001b[22mloading\u001b[2m \u2502\u001b[6;1H\u2570\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u256f\u001b[21;1H\u001b[22m\u001b[1m\u203a\u001b[21;3H\u001b[22m\u001b[2m\u001b[2mAsk Codex to do anything\u001b[24;3H\u001b[22m\u001b[1m?\u001b[24;5H\u001b[22mfor\u001b[24;9Hshortcuts\u001b[39m\u001b[49m\u001b[0m\u001b[1;1H\u001b[0 q\u001b[1;1H\u001b[2m\u256d\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l\u001b[?2026l" +}, +{ +"t": 303, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 304, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 305, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 306, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 307, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 364, +"dir": "out", +"data": "\u001b[?2026h\u001b[?25l\u001b[1;41H\u001b[2m\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u256e\u001b[2;41H \u2502\u001b[3;41H \u2502\u001b[4;41H \u2502\u001b[5;14H\u001b[22m/private/tmp/claude-501/\u2026/rec/untrusted-ABCD\u001b[2m \u2502\u001b[6;41H\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u256f\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h" +}, +{ +"t": 364, +"dir": "out", +"data": "\u001b[?2026l" +}, +{ +"t": 365, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 365, +"dir": "out", +"data": "\u001b[?1006l\u001b[?1015l\u001b[?1003l\u001b[?1002l\u001b[?1000l\u001b[<1u\u001b[?1007l\u001b[?1049l\u001b[<1u\u001b[>4;0m\u001b[?2004l\u001b[?1004l\u001b[0 q\u001b[?25h" +}, +{ +"t": 1522, +"dir": "out", +"data": "\u001b[?2004h\u001b[>4;0m\u001b[>7u\u001b[?1004h\u001b[?25l\u001b[?1049h\u001b[>4;0m\u001b[>7u\u001b[?1007l\u001b[?1000h\u001b[?1002h\u001b[?1006h\u001b[?1003h\u001b[?25l\u001b[1;1H\u001b[J\u001b[?2026h\u001b[?25l\u001b[1;1H\u001b[2m\u256d\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u256e\u001b[2;1H\u2502 >_ \u001b[22m\u001b[1mOpenAI Codex\u001b[22m\u001b[2m\u001b[2m (v0.157.1) \u2502\u001b[3;1H\u2502 \u2502\u001b[4;1H\u2502 model: \u001b[3mloading\u001b[23m \u001b[22m\u001b[38;2;99;168;248;49m/model\u001b[2m\u001b[39;49m to change \u2502\u001b[5;1H\u2502 directory: \u001b[22m/private/tmp/claude-501/\u2026/rec/untrusted-ABCD\u001b[2m \u2502\u001b[6;1H\u2570\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u256f\u001b[21;1H\u001b[22m\u001b[1m\u203a\u001b[21;3H\u001b[22m\u001b[2m\u001b[2mAsk Codex to do anything\u001b[24;3H\u001b[22m\u001b[1m?\u001b[24;5H\u001b[22mfor\u001b[24;9Hshortcuts\u001b[39m\u001b[49m\u001b[0m\u001b[1;1H\u001b[0 q\u001b[1;1H\u001b[2m\u256d\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25" +}, +{ +"t": 1522, +"dir": "out", +"data": "h\u001b[?2026l" +}, +{ +"t": 1530, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1637, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1691, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1694, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1716, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1742, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1742, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1744, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1745, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1753, +"dir": "out", +"data": "\u001b[?2026h\u001b[39m\u001b[49m\u001b[0m\u001b[21;3H\u001b[?25h\u001b[?2026l" +}, +{ +"t": 1848, +"dir": "out", +"data": "\u001b[?2026h\u001b[?25l" +}, +{ +"t": 1848, +"dir": "out", +"data": "\u001b[1;2H\u001b[0m\u001b[49m\u001b[K\u001b[2;16H\u001b[0m\u001b[49m\u001b[K\u001b[5;2H\u001b[0m\u001b[49m\u001b[K\u001b[21;2H\u001b[0m\u001b[49m\u001b[K\u001b[24;2H\u001b[0m\u001b[49m\u001b[K\u001b[1;1H \u001b[2;1H\u001b[1m Folder access\u001b[3;1H\u001b[22m \u001b[2m/private/tmp/claude-501/-Users-fixture-user-Desktop-Development-agent-code/d\u001b[4;1H\u001b[22m \u001b[2m0000000-0000-0000-0000-000000000000/scratchpad/rec/untrusted-ABCD \u001b[5;1H\u001b[22m \u001b[6;1H Trust this folder? Codex can read, edit, and run files here,\u001b[6;64Hsubject\u001b[6;72Hto\u001b[6;75Hyour\u001b[7;3Hpermission\u001b[7;14Hsettings.\u001b[7;24HFolder\u001b[7;31Hsettings\u001b[7;40Hcan\u001b[7;44Hrun\u001b[7;48Hcode\u001b[7;53Hautomatically,\u001b[7;68Heven\u001b[8;3Hwithout\u001b[8;11Ha\u001b[8;13Hmodel\u001b[8;19Hrequest.\u001b[8;28HContinue\u001b[8;37Honly\u001b[8;42Hif\u001b[8;45Hyou\u001b[8;49Htrust\u001b[8;55Hthese\u001b[8;61Hfiles.\u001b[8;68HYour\u001b[8;73Htrust\u001b[9;3Hdecision\u001b[9;12Hwill\u001b[9;17Hbe\u001b[9;20Hsaved.\u001b[11;1H\u001b[7m\u001b[1m\u203a 1. Trust and continue \u001b[12;3H\u001b[27m\u001b[22m2.\u001b[12;6HBack\u001b[12;11Hto\u001b[12;14HAgent\u001b[12;20HCommand\u001b[12;28HCenter\u001b[14;3H\u001b[1menter\u001b[22m\u001b[2m\u001b[2m continue \u00b7 \u001b[22m\u001b[1mesc\u001b[22m\u001b[2m\u001b[2m back\u001b[21;1H\u001b[22m \u001b[39m\u001b[49m\u001b[" +}, +{ +"t": 1848, +"dir": "out", +"data": "0m\u001b[?2026l" +}, +{ +"t": 9010, +"dir": "in", +"label": "observed", +"data": "" +}, +{ +"t": 9019, +"dir": "exit", +"code": 0 +} +] +} \ No newline at end of file diff --git a/testing/fixtures/trust-dialog-0157/upstream-snapshots-0157.1.json b/testing/fixtures/trust-dialog-0157/upstream-snapshots-0157.1.json new file mode 100644 index 0000000..09e885f --- /dev/null +++ b/testing/fixtures/trust-dialog-0157/upstream-snapshots-0157.1.json @@ -0,0 +1,50 @@ +{ + "evidence": "Upstream insta snapshots of TrustDirectoryWidget at tag rust-v0.157.1 (commit 36650394c5b3), copied verbatim from openai/codex (Apache-2.0). They are upstream TEST OUTPUT, not local recordings: they cover the layout variants a single local folder cannot produce (Git subdirectory, saved-untrusted, existing task, trust-write error, 40-column truncation).", + "snapshots": [ + { + "name": "existing_untrusted_task", + "source": "openai/codex rust-v0.157.1 codex-rs/tui/src/onboarding/snapshots/codex_tui__onboarding__trust_directory__tests__existing_untrusted_task.snap", + "frame": "\n Folder access\n /workspace/project \n\n This existing task may retain settings and history, including\n project configuration or hooks loaded while it was trusted. To use\n restricted settings, start a new task. The folder's trust setting\n will not change.\n\n› 1. Open existing task \n 2. Back to Agent Command Center\n\n enter continue · esc back" + }, + { + "name": "folder_picker_restricted_40x24", + "source": "openai/codex rust-v0.157.1 codex-rs/tui/src/onboarding/snapshots/codex_tui__onboarding__trust_directory__tests__folder_picker_restricted_40x24.snap", + "frame": "\n Folder access\n /workspace/project/long-nested-folde\n r \n\n Config, hooks, and exec policies\n from untrusted folders stay\n disabled. Trusted project folders\n can still contribute settings.\n Skills still load, and tools follow\n your permission settings. Opening\n will not change saved trust.\n\n› 1. Open restricted \n 2. Back to Agent Command Center\n\n enter continue · esc back" + }, + { + "name": "long_checkout_40x13", + "source": "openai/codex rust-v0.157.1 codex-rs/tui/src/onboarding/snapshots/codex_tui__onboarding__trust_directory__tests__long_checkout_40x13.snap", + "frame": " Folder access\n workspace/…/repository \n Trust this folder? Codex can read,\n edit, and run files here, subject to\n your permission settings. Folder\n settings can run code automatically,\n even without a model request.\n Continue only if you trust these\n files. Your trust decision will be\n saved.\n› 1. Trust and continue \n 2. Quit\n enter continue · esc quit" + }, + { + "name": "long_repository_root_40x17", + "source": "openai/codex rust-v0.157.1 codex-rs/tui/src/onboarding/snapshots/codex_tui__onboarding__trust_directory__tests__long_repository_root_40x17.snap", + "frame": " Folder access\n workspace/…/repository/checkout \n Note: You’re in a subdirectory of a \n Git project. Trusting will apply to \n the repository root: \n workspace/…/repository \n Trust this folder? Codex can read,\n edit, and run files here, subject to\n your permission settings. Folder\n settings can run code automatically,\n even without a model request.\n Continue only if you trust these\n files. Your trust decision will be\n saved.\n› 1. Trust and continue \n 2. Quit\n enter continue · esc quit" + }, + { + "name": "only_repository_root_fits_40x16", + "source": "openai/codex rust-v0.157.1 codex-rs/tui/src/onboarding/snapshots/codex_tui__onboarding__trust_directory__tests__only_repository_root_fits_40x16.snap", + "frame": " Folder access\n Note: You’re in a subdirectory of a \n Git project. Trusting will apply to \n the repository root: \n workspace/…/repository \n Trust this folder? Codex can read,\n edit, and run files here, subject to\n your permission settings. Folder\n settings can run code automatically,\n even without a model request.\n Continue only if you trust these\n files. Your trust decision will be\n saved.\n› 1. Trust and continue \n 2. Quit\n enter continue · esc quit" + }, + { + "name": "renders_restricted_folder", + "source": "openai/codex rust-v0.157.1 codex-rs/tui/src/onboarding/snapshots/codex_tui__onboarding__trust_directory__tests__renders_restricted_folder.snap", + "frame": "\n Folder access\n /workspace/project \n\n Config, hooks, and exec policies from untrusted folders stay\n disabled. Trusted project folders can still contribute settings.\n Skills still load, and tools follow your permission settings.\n Opening will not change saved trust.\n\n› 1. Open restricted \n 2. Back to Agent Command Center\n\n enter continue · esc back" + }, + { + "name": "renders_snapshot_for_git_repo", + "source": "openai/codex rust-v0.157.1 codex-rs/tui/src/onboarding/snapshots/codex_tui__onboarding__trust_directory__tests__renders_snapshot_for_git_repo.snap", + "frame": "\n Folder access\n /workspace/project \n\n Trust this folder? Codex can read, edit, and run files here,\n subject to your permission settings. Folder settings can run code\n automatically, even without a model request. Continue only if you\n trust these files. Your trust decision will be saved.\n\n› 1. Trust and continue \n 2. Quit\n\n enter continue and create sandbox · esc quit" + }, + { + "name": "renders_snapshot_for_remote_git_subdirectory", + "source": "openai/codex rust-v0.157.1 codex-rs/tui/src/onboarding/snapshots/codex_tui__onboarding__trust_directory__tests__renders_snapshot_for_remote_git_subdirectory.snap", + "frame": "\n Folder access\n /srv/remote/project/nested\n\n Note: You’re in a subdirectory of a Git project. Trusting will\n apply to the repository root:\n /srv/remote/project\n\n Trust this folder? Codex can read, edit, and run files here,\n subject to your permission settings. Folder settings can run code\n automatically, even without a model request. Continue only if you\n trust these files. Your trust decision will be saved.\n\n› 1. Trust and continue\n 2. Back to Agent Command Center\n\n enter continue · esc back" + }, + { + "name": "renders_snapshot_for_trust_error", + "source": "openai/codex rust-v0.157.1 codex-rs/tui/src/onboarding/snapshots/codex_tui__onboarding__trust_directory__tests__renders_snapshot_for_trust_error.snap", + "frame": "\n Folder access\n /workspace/project \n\n Trust this folder? Codex can read, edit, and run files here,\n subject to your permission settings. Folder settings can run code\n automatically, even without a model request. Continue only if you\n trust these files. Your trust decision will be saved.\n\n› 1. Trust and continue \n 2. Quit\n\n Failed to set trust for /workspace/project: config/batchWrite \n failed in TUI: Invalid configuration: features.fast_mode=true is \n not supported; allowed set [fast_mode=false] \n\n enter continue · esc quit" + } + ] +} \ No newline at end of file diff --git a/testing/record-live-config-read.mts b/testing/record-live-config-read.mts index 37d2a89..9c32d22 100644 --- a/testing/record-live-config-read.mts +++ b/testing/record-live-config-read.mts @@ -16,10 +16,13 @@ const INITIALIZE_ID = 'agent-code-config-initialize' const CONFIG_READ_ID = 'agent-code-config-read' const binary = process.env.CODEX_BINARY ?? 'codex' const cwd = process.env.CODEX_INPUT_RECORD_CWD ?? process.cwd() -const sourceEvidencePath = fileURLToPath(new URL( - './fixtures/prompt-input/codex-01491-config-source.json', - import.meta.url, -)) +// WHY a table of exact recorded versions (#63): the upstream commit is not +// derivable from the binary, and the source evidence is a per-tag audit of the +// config precedence code. A version with no row here has no audited source. +const RECORDED_UPSTREAM: Record = { + '0.149.1': { commit: 'ff29a44391deccde0aba0f8390337d7f3c319ea4', sourceEvidence: 'codex-01491-config-source.json' }, + '0.157.1': { commit: '36650394c5b38c2990ccf2a3457165ca3e9d9726', sourceEvidence: 'codex-01571-config-source.json' }, +} // WHY this recorder projects only key routing, version, and layer types. A // config/read response may contain private paths, MCP declarations, and future @@ -27,13 +30,19 @@ const sourceEvidencePath = fileURLToPath(new URL( // fixture into a secret/configuration archive. The production attestor must // likewise inspect in memory and return only a capability or a generic refusal. const projection = await recordProjection() +const upstream = RECORDED_UPSTREAM[projection.cliVersion] +if (!upstream) throw new Error(`no audited upstream source for Codex ${projection.cliVersion}`) +const sourceEvidencePath = fileURLToPath(new URL( + `./fixtures/prompt-input/${upstream.sourceEvidence}`, + import.meta.url, +)) process.stdout.write(`${JSON.stringify({ schemaVersion: 1, provider: { cliVersion: projection.cliVersion, binarySha256: sha256File(binary), - upstreamTag: 'rust-v0.149.1', - upstreamCommitSha: 'ff29a44391deccde0aba0f8390337d7f3c319ea4', + upstreamTag: `rust-v${projection.cliVersion}`, + upstreamCommitSha: upstream.commit, }, protocol: { transport: 'app-server-stdio-jsonl', diff --git a/testing/record-live-prompt-input.mts b/testing/record-live-prompt-input.mts index 122f2b3..807b45a 100644 --- a/testing/record-live-prompt-input.mts +++ b/testing/record-live-prompt-input.mts @@ -30,6 +30,10 @@ type InputCase = { waitBeforeFinal?: RegExp setup?: (session: LiveSession) => Promise> afterDraft?: (session: LiveSession) => Promise> + /** Files created in the workspace before launch (e.g. for `@` file search). */ + workspaceFiles?: Record + /** Record only on CLIs that support the 0.156+ adjustments (see cliSupportsNoDaemon). */ + only0156Plus?: boolean } type CapturedRequest = { @@ -78,11 +82,18 @@ const fixtureSse = [ '', '', ].join('\n') +const TITLE_REQUEST_PREFIX = 'Generate a concise, single-line task title' const binary = await readFile(CODEX_BINARY) const binarySha256 = sha256(binary) const cliVersion = await binaryVersion() const requests: CapturedRequest[] = [] -let releaseSlowResponse: (() => void) | null = null +// WHY a list, not one slot (#63): the first 0.157.1 diagnosis was a second +// held slow-turn request overwriting this slot. That second request turned out +// to be the thread-title side request (now answered unheld and unrecorded in +// serveFixture). A list still releases every held request, so no future +// side request can strand the turn the same way. +const slowResponseReleasers: Array<() => void> = [] +const releaseSlowResponses = (): void => { for (const release of slowResponseReleasers.splice(0)) release() } const server = createServer(async (req, res) => { await serveFixture(req, res) @@ -109,13 +120,28 @@ const allCases: InputCase[] = [ expectedDurableText: 'RECORDED_TRUST_PROMPT', setup: async session => { await waitForScreen(session, screen => - screen.includes('Do you trust the contents of this directory'), + screen.includes('Do you trust the contents of this directory') || + screen.includes('1. Trust and continue'), ) const modal = structuralScreen(session) await delay(300) + if (!session.mirror.snapshotPlain().includes('1. Trust and continue')) { + session.terminal.write('1') + await waitForComposer(session) + return { nonComposerWrites: ['1'], modal } + } + // WHY 0.156+ takes '1' then Enter (#63, codex-headless#65): upstream + // changed `1` to move the highlight only; Enter confirms. Recorded here + // in the isolated CODEX_HOME rather than assumed: after '1' alone the + // dialog must still be up, and only the Enter reaches the composer. session.terminal.write('1') + await delay(1000) + if (!session.mirror.snapshotPlain().includes('1. Trust and continue')) { + throw new Error("0.156+ trust dialog closed on '1' alone; the recorded key contract changed") + } + session.terminal.write('\r') await waitForComposer(session) - return { nonComposerWrites: ['1'], modal } + return { nonComposerWrites: ['1', '\r'], modal } }, }, { @@ -157,7 +183,12 @@ const allCases: InputCase[] = [ configOverrides: ['tui.vim_mode_default=true'], inputChunks: ['i', 'abc', '\r'], expectedSubmission: true, - expectedDurableText: 'abc', + // WHY version-specific (#63): 0.149.1 opened a Vim-default composer in + // Normal mode, so the leading `i` only entered Insert. rust-v0.157.1's + // chatwidget/constructor.rs calls `enable_vim_in_insert_mode()` instead, + // so the same `i` is typed text. The case id keeps its historical name so + // the two corpora stay comparable row by row. + expectedDurableText: cliStartsVimInInsert(cliVersion) ? 'iabc' : 'abc', }, { id: 'unbound-submit-enter', @@ -177,7 +208,12 @@ const allCases: InputCase[] = [ id: 'tab-footer-spoof-skill-popup', inputChunks: ['literal tab to queue $recorded_evi', '\t'], expectedSubmission: false, - waitForPopup: /Plugin|Skill|App/i, + // WHY the popup's own key hint (#63): the skill list for `$recorded_evi` + // reads "no matches" in both recorded versions, so row words like "Skill" + // only matched a transient loading frame and made the wait flaky. The hint + // is the popup: 0.149.1 paints "Press enter to insert or esc to close" + // below the composer; 0.157.1 paints "enter insert · esc close" above it. + waitForPopup: /Press enter to insert or esc to close|enter insert · esc close/i, }, { id: 'active-footer-tab-queue', @@ -267,13 +303,36 @@ const allCases: InputCase[] = [ startupOnly: true, expectedStartupFailure: /composer\.queue.*composer\.toggle_shortcuts|composer\.toggle_shortcuts.*composer\.queue/i, }, + // WHY these two (review a of codex-headless#69): 0.157.1 paints EVERY popup + // above the composer, and the slash-command and file popups carry no hint + // row. A frame with a popup open therefore looks like an idle composer with + // a draft, and Enter, which selects the popup's item, must not count as a + // submission of the draft. Enter is the key the skill-popup case never + // pressed (it pressed Tab). Recorded on 0.156+ only: the 0.149.1 corpus + // cannot be re-recorded, because that binary is no longer installed. + { + id: 'slash-popup-enter-selects-command', + only0156Plus: true, + inputChunks: ['/stat', '\r'], + expectedSubmission: false, + waitBeforeFinal: /\/status/, + }, + { + id: 'file-popup-enter-inserts-mention', + only0156Plus: true, + workspaceFiles: { 'README.md': '# recorded fixture\n' }, + inputChunks: ['@READ', '\r'], + expectedSubmission: false, + waitBeforeFinal: /README\.md/, + }, ] const requestedCases = new Set( (process.env.CODEX_INPUT_RECORD_CASES ?? '').split(',').filter(Boolean), ) +const versionCases = allCases.filter(inputCase => !inputCase.only0156Plus || cliSupportsNoDaemon(cliVersion)) const cases = requestedCases.size === 0 - ? allCases - : allCases.filter(inputCase => requestedCases.has(inputCase.id)) + ? versionCases + : versionCases.filter(inputCase => requestedCases.has(inputCase.id)) const output: Record[] = [] try { @@ -416,12 +475,12 @@ try { screen.includes('Queued follow-up inputs') && screen.includes('RECORDED_QUEUED_PROMPT'), ) - releaseSlowResponse?.() - releaseSlowResponse = null + releaseSlowResponses() } let durableUserText: string | null = null let requestUserText: string | null = null + let promptRequest: CapturedRequest | undefined if (inputCase.expectedSubmission) { try { await waitFor(async () => { @@ -447,11 +506,19 @@ try { `last screen=${JSON.stringify(structuralScreen(session))}`, ) } - await waitFor(() => session.requests.length > beforeRequests, + // WHY find the request that carries the prompt, not the last one + // (#63): 0.157.1 can re-send a request that the fixture holds (the + // slow turn) after its own idle timeout. So after a queued follow-up, + // the newest request is sometimes that retry, not the follow-up. The + // agreement still has to hold: a request whose final user message is + // the durable prompt must arrive, or the recording fails. + const matchingRequest = () => session.requests.slice(beforeRequests) + .filter(request => extractLastRequestUserText(request.body) === durableUserText) + .at(-1) + await waitFor(() => matchingRequest() !== undefined, `${inputCase.id} fixture request`) - requestUserText = extractLastRequestUserText( - session.requests.at(-1)?.body, - ) + promptRequest = matchingRequest()! + requestUserText = extractLastRequestUserText(promptRequest.body) } else { await delay(1_200) const afterUsers = await readDurableUserTexts(session.codexHome) @@ -473,8 +540,8 @@ try { const rawPtySha256 = sha256(session.rawPtyChunks.join('')) const rolloutSha256 = await hashRolloutCorpus(session.codexHome) - const rawRequestSha256 = session.requests.length > beforeRequests - ? sha256(JSON.stringify(session.requests.at(-1)!.body)) + const rawRequestSha256 = promptRequest + ? sha256(JSON.stringify(promptRequest.body)) : null output.push({ id: inputCase.id, @@ -503,8 +570,7 @@ try { ...afterDraft, }) } finally { - releaseSlowResponse?.() - releaseSlowResponse = null + releaseSlowResponses() try { session.terminal.kill() } catch { /* Provider may already have exited. */ } session.mirror.dispose() await rm(session.codexHome, { recursive: true, force: true }) @@ -521,7 +587,9 @@ process.stdout.write(`${JSON.stringify({ provider: { cliVersion, binarySha256, - upstreamTag: 'rust-v0.149.1', + // Derived, not a literal: the 0.157.1 corpus (#63) is recorded by the + // same script, and a hard-coded 0.149.1 tag mislabelled its provenance. + upstreamTag: `rust-v${/(\d+\.\d+\.\d+)/.exec(cliVersion)?.[1] ?? 'unknown'}`, }, terminal: { cols: COLS, rows: ROWS }, source: { @@ -542,9 +610,20 @@ async function startSession(inputCase: InputCase): Promise { ? join(workspaceRoot, inputCase.workspaceSuffix) : workspaceRoot if (inputCase.workspaceSuffix) await mkdir(workspace, { recursive: true }) + for (const [name, content] of Object.entries(inputCase.workspaceFiles ?? {})) { + await writeFile(join(workspace, name), content) + } await mkdir(join(codexHome, 'skills', 'recorded-evidence'), { recursive: true }) await writeFile( join(codexHome, 'skills', 'recorded-evidence', 'SKILL.md'), + // WHY frontmatter from 0.156 (#63): 0.157.1 rejects a SKILL.md without + // YAML frontmatter ("missing YAML frontmatter delimited by ---"). The skill + // then does not load, so the `$` popup case has nothing to pop, and the + // startup warning takes over the footer row where `reverse-i-search:` + // paints. Gated so the 0.149.1 recording's skill file is unchanged. + (cliSupportsNoDaemon(cliVersion) + ? '---\nname: recorded-evidence\ndescription: A harmless local fixture used only by the prompt-input recorder.\n---\n\n' + : '') + '# Recorded evidence\n\nA harmless local fixture used only by the prompt-input recorder.\n', ) const trusted = inputCase.trusted ?? true @@ -564,6 +643,17 @@ async function startSession(inputCase: InputCase): Promise { 'trust_level = "trusted"', '', ] : []), + // WHY from 0.156 (#63): 0.157.1's bundled models.json lists + // gpt-5.6-sol -> gpt-6-sol as an upgrade, so a fresh CODEX_HOME opens a + // "Try new model" startup prompt before the composer. Recording that the + // migration was already seen (the same table Codex writes when the user + // answers) keeps the recorded model identical to the 0.149.1 corpus. Gated + // so the 0.149.1 recording's config stays byte-for-byte what it was. + ...(cliSupportsNoDaemon(cliVersion) ? [ + '[notice.model_migrations]', + '"gpt-5.6-sol" = "gpt-6-sol"', + '', + ] : []), ...(inputCase.lowerLayerConfig ?? []), ...(inputCase.lowerLayerConfig?.length ? [''] : []), ].join('\n') @@ -572,7 +662,21 @@ async function startSession(inputCase: InputCase): Promise { const args = [ '--sandbox', 'read-only', '--ask-for-approval', 'never', - '--no-alt-screen', + // WHY optional from 0.157 (#63): 0.157 made the fullscreen transcript + // (alternate screen) the default, and Agent Code launches Codex WITHOUT + // this flag. A corpus recorded only inline would not describe the screen + // the app's panes actually show. CODEX_INPUT_RECORD_ALT_SCREEN=1 records + // with the app's launch shape; the default keeps the historical inline + // recording comparable with the 0.149.1 corpus. + ...(process.env.CODEX_INPUT_RECORD_ALT_SCREEN === '1' ? [] : ['--no-alt-screen']), + // WHY --no-daemon from 0.157 (#63, #66): 0.157 auto-starts a managed + // app-server daemon whose control socket lives under CODEX_HOME. This + // recorder's isolated CODEX_HOME under macOS $TMPDIR makes that socket + // path longer than SUN_LEN, and Codex then refuses to start at all ("app + // server did not become ready … path must be shorter than SUN_LEN"). The + // daemon it spawned also outlives the deleted home. The flag is gated + // because older CLIs reject unknown arguments. + ...(cliSupportsNoDaemon(cliVersion) ? ['--no-daemon'] : []), ] for (const override of inputCase.configOverrides ?? []) { args.push('-c', override) @@ -624,6 +728,18 @@ async function serveFixture(req: IncomingMessage, res: ServerResponse): Promise< for await (const chunk of req) chunks.push(Buffer.from(chunk)) const bodyText = Buffer.concat(chunks).toString('utf8') const body = JSON.parse(bodyText) as unknown + // WHY title requests are answered but never recorded or held (#63). 0.157.1 + // sends a SECOND request per first prompt, a thread-title side request + // (tui/src/app/thread_title.rs at rust-v0.157.1: "Generate a concise, + // single-line task title … User prompt:\n"). It is not the user's + // turn. Recorded, it became `requests.at(-1)` and the rollout/request + // agreement compared the prompt against the title prompt. Held, because it + // quotes RECORDED_SLOW_TURN, it kept the slow turn's queue from draining. + if (extractLastRequestUserText(body)?.startsWith(TITLE_REQUEST_PREFIX)) { + res.writeHead(200, { 'content-type': 'text/event-stream', 'cache-control': 'no-cache', connection: 'keep-alive' }) + res.end(fixtureSse) + return + } requests.push({ body, receivedAt: Date.now() }) if (bodyText.includes('RECORDED_SLOW_TURN')) { // Keep the HTTP request pending before any SSE bytes are written. Splitting @@ -631,7 +747,7 @@ async function serveFixture(req: IncomingMessage, res: ServerResponse): Promise< // report an incomplete stream; holding the request itself preserves the // provider's real working/queue UI while the eventual response remains the // same independently validated complete fixture used by the control turns. - await new Promise(resolveRelease => { releaseSlowResponse = resolveRelease }) + await new Promise(resolveRelease => { slowResponseReleasers.push(resolveRelease) }) } res.writeHead(200, { 'content-type': 'text/event-stream', @@ -683,8 +799,17 @@ async function waitFor( function structuralScreen(session: LiveSession): string[] { const screenRows = session.mirror.snapshotPlain().split('\n') - const rowCount = screenRows.some(row => row.includes('Do you trust the contents')) ? 12 : 8 - return screenRows.slice(-rowCount).map(row => sanitizeScreenRow(session, row)) + const rowCount = screenRows.some(row => row.includes('Do you trust the contents')) ? 12 + : screenRows.some(row => row.includes('1. Trust and continue')) ? 16 : 8 + // WHY end at the last painted row from 0.156 (#63): 0.157.1 fullscreen + // paints its trust dialog (and a short session) from the TOP of the screen, + // so a bottom window recorded only blank rows. Gated so the 0.149.1 + // recording conditions are unchanged. + let end = screenRows.length + if (cliSupportsNoDaemon(cliVersion)) { + while (end > 0 && screenRows[end - 1]!.trim() === '') end-- + } + return screenRows.slice(Math.max(0, end - rowCount), end).map(row => sanitizeScreenRow(session, row)) } function sanitizeScreenRow(session: LiveSession, row: string): string { @@ -740,9 +865,18 @@ async function recordResizeBoundary( throw new Error('provider bytes arrived inside the synchronous resize boundary') } + // WHY wait for the rows to CHANGE, not merely for newer bytes (#63). In + // 0.157.1 fullscreen the first bytes after SIGWINCH are a status/spinner + // chunk, followed by a quiet gap, and the composer is repainted at the new + // width only later, on the TUI's frame timer (live: still the 52-column wrap + // at 300 ms, repainted by 3.3 s). Stopping at the first quiet point recorded + // a stale frame as "after provider redraw", and the replay test then + // projected a repaint onto it that never happened. + const narrowRows = narrow.rows.map(row => row.text).join('\n') await waitFor(() => { const frame = session.mirror.snapshotStableFrame() - return frame !== null && frame.generation > beforeProviderRedraw.generation + return frame !== null && frame.generation > beforeProviderRedraw.generation && + frame.rows.map(row => row.text).join('\n') !== narrowRows }, 'provider redraw after resize') await waitForPtyQuiet(session) const afterProviderRedraw = session.mirror.snapshotStableFrame() @@ -863,8 +997,7 @@ async function recordUnchangedRedrawAfterEdit( throw new Error('slow setup prompt did not agree at rollout/request boundaries') } - releaseSlowResponse?.() - releaseSlowResponse = null + releaseSlowResponses() await waitForScreen(session, screen => !screen.includes('Working (') && screen.includes(edit), ) @@ -959,10 +1092,15 @@ async function waitForPtyQuiet(session: LiveSession, quietMs = 200): Promise { const start = Math.max(0, end - 10) return { @@ -1051,6 +1189,22 @@ async function hashRolloutCorpus(codexHome: string): Promise { return hash.digest('hex') } +// Exact recorded versions only: the Normal-to-Insert start changed somewhere +// between 0.149.1 and 0.157.1, and no intermediate release was recorded. +function cliStartsVimInInsert(version: string): boolean { + return /\b0\.157\.1\b/.test(version) +} + +// Also the version gate for the 0.156+ recording adjustments in startSession. +// `--no-daemon` first appears in codex --help at 0.156 (checked against the +// installed 0.150.1–0.157.1 standalone releases: absent through 0.155.1, present from 0.156.0). +function cliSupportsNoDaemon(version: string): boolean { + const match = /(\d+)\.(\d+)\.(\d+)/.exec(version) + if (!match) return false + const [major, minor] = [Number(match[1]), Number(match[2])] + return major > 0 || minor >= 156 +} + async function binaryVersion(): Promise { const child = pty.spawn(CODEX_BINARY, ['--version'], { name: 'xterm-256color', cols: 80, rows: 10, cwd: process.cwd(),