From 60d926d9a2872d96f92b5f8eeed4c85894d351a6 Mon Sep 17 00:00:00 2001 From: danielkuang Date: Sat, 19 Sep 2026 13:27:35 +0800 Subject: [PATCH] fix: reject DeepSeek WebSocket upgrades with 426 for a zero-retry HTTP fallback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The client treats a DeepSeek model on an accepted socket as a retryable stream error and burns 5 reconnect attempts (~10-15s of backoff) before falling back to HTTP — the case the v1.2.1 release notes flagged as unverified in normal DeepSeek use. The handshake already carries the model in x-codex-routing-hint, and the client maps an upgrade-time HTTP 426 directly to its no-retry HTTP fallback (WebsocketStreamOutcome:: FallbackToHttp), so reject DeepSeek-hinted upgrades early. The first-frame close-1008 path stays as the fallback for handshakes without the hint. AGENTS.md rule 20 and both READMEs are updated. --- AGENTS.md | 8 ++++-- README.en.md | 2 +- README.md | 2 +- src/websocket-proxy.mjs | 20 +++++++++++++ test/proxy.test.mjs | 63 ++++++++++++++++++++++++++++++++++++++++- 5 files changed, 90 insertions(+), 5 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 2b5c8f5..153ab43 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -141,5 +141,9 @@ The non-negotiable details: 20. Pets, plugins, skills, and MCP remain client-side. Voice uses GPT-Live and must never route to DeepSeek; Realtime routing compatibility is still pending PR #21 validation. DeepSeek catalog entries keep `prefer_websockets = false`. Native GPT entries keep `prefer_websockets = true`. - Authorized `/v1/responses` WebSocket upgrades are proxied to chatgpt.com; a DeepSeek model on - that socket is closed so the client falls back to HTTP. Other upgrade probes still receive 426. + Authorized `/v1/responses` WebSocket upgrades are proxied to chatgpt.com. When the handshake + carries a DeepSeek model in `x-codex-routing-hint` (ChatGPT-auth clients send + `model=;tier=`), reject the upgrade with HTTP 426: the client maps that response to + a direct HTTP fallback with no reconnect retries. When the hint is missing, a DeepSeek model on + the accepted socket is still closed with 1008 to force the same fallback. Other upgrade probes + still receive 426. diff --git a/README.en.md b/README.en.md index 7c1df2b..bd7fae0 100644 --- a/README.en.md +++ b/README.en.md @@ -166,7 +166,7 @@ Yes. Tool calls and web search go through DeepSeek's Responses API. Flash handle - **Reasoning folds mid-task.** DeepSeek emits `response.completed` after every tool round; Codex folds the reasoning block, runs the tool, and opens the next round. This is API behavior, not a bug. Tool-free turns fold once at the end. - **Native vision.** Images and tool-returned images go straight to `deepseek-flash`; GPT image descriptions and `DSCODEX_VISION_MODEL` are no longer used. - **DeepSeek → GPT thread history.** Before forwarding to GPT the router strips foreign plaintext reasoning and restores its own encrypted compaction summary as assistant context. Native GPT reasoning and ordinary requests keep their original bytes; rollout files are never rewritten. The same rewrite runs on HTTP SSE and on every Responses WebSocket `response.create`. -- **Official GPT WebSocket.** Desktop 26.908+ dials the loopback WS first. The router has to be running for that upgrade to reach chatgpt.com; if it is down, official models Reconnecting 5/5. A DeepSeek model on the same socket is closed with 1008 so the client falls back to HTTP Responses. +- **Official GPT WebSocket.** Desktop 26.908+ dials the loopback WS first. The router has to be running for that upgrade to reach chatgpt.com; if it is down, official models Reconnecting 5/5. A DeepSeek-hinted handshake (always present with ChatGPT auth) is rejected with HTTP 426 so the client switches to HTTP Responses with no reconnect retries; without the hint a DeepSeek model on the socket is closed with 1008 to force the same fallback. - **Voice.** GPT-Live is never sent to DeepSeek. Realtime routing compatibility is pending PR #21 validation. Pets, plugins, skills and MCP remain client-side. - **Acceptance scope.** CI covers macOS / Windows / Linux. Windows on real hardware (desktop app plus autostart) has not been accepted on the maintainer's machine. - **Key storage, proxy resolution, bridge details, platform differences.** See `AGENTS.md`. diff --git a/README.md b/README.md index fb66124..d4af01f 100644 --- a/README.md +++ b/README.md @@ -167,7 +167,7 @@ ChatGPT 桌面端 26.908+ 会先连 `ws://127.0.0.1:10110//v1/responses` - **思考块反复折叠。** DeepSeek 每轮工具调用结束都发 `response.completed`,Codex 随之折叠思考、执行工具、再展开下一轮。这是 API 行为,不是 bug;无工具的单轮只折叠一次。 - **原生识图。** 图片和工具返回的图片直接交给 `deepseek-flash`,不再借 GPT 代读;`DSCODEX_VISION_MODEL` 不再生效。 - **DeepSeek → GPT 任务历史。** 转发 GPT 前剥掉外来明文 reasoning,把 DSCodex 自己的加密压缩摘要恢复为助手上下文;GPT 原生 reasoning 与普通请求保持原始字节,rollout 文件不改写。HTTP SSE 与每条 Responses WebSocket `response.create` 都做这件事。 -- **官方 GPT WebSocket。** 桌面端 26.908+ 先连 loopback WS。路由器必须在跑,upgrade 才会透传到 chatgpt.com;停掉就 Reconnecting 5/5。DeepSeek 若被打到同一条 WS,close 1008 后回 HTTP Responses。 +- **官方 GPT WebSocket。** 桌面端 26.908+ 先连 loopback WS。路由器必须在跑,upgrade 才会透传到 chatgpt.com;停掉就 Reconnecting 5/5。DeepSeek 的 WS 握手带模型提示(ChatGPT 登录始终会带)时会被直接拒绝(HTTP 426),客户端零重试切到 HTTP Responses;提示缺失时仍在首帧按 close 1008 回退。 - **Voice。** GPT-Live 从不发给 DeepSeek;Realtime 路由兼容仍待 PR #21 验收。Pets、插件、技能与 MCP 仍由客户端处理。 - **验收范围。** CI 覆盖 macOS / Windows / Linux;Windows 实机(桌面端 + 自启动)尚未在维护者机器上验收。 - **Key 存储、代理解析、bridge 细节、平台差异。** 见 `AGENTS.md`。 diff --git a/src/websocket-proxy.mjs b/src/websocket-proxy.mjs index 2bca4ac..0fe646f 100644 --- a/src/websocket-proxy.mjs +++ b/src/websocket-proxy.mjs @@ -12,6 +12,19 @@ export function rejectUpgrade(socket, statusLine = "426 Upgrade Required") { socket.end(`HTTP/1.1 ${statusLine}\r\nConnection: close\r\n\r\n`); } +// The Codex client sends `x-codex-routing-hint: model=;tier=` on the +// WebSocket handshake (ChatGPT-auth sessions). Rejecting a DeepSeek upgrade with +// HTTP 426 makes the client fall back to HTTP immediately via +// `WebsocketStreamOutcome::FallbackToHttp`, with no reconnect retries — instead +// of accepting the socket and closing on the first frame, which costs 5 backoff +// retries (~10-15s) per thread. +export function routingHintModel(request) { + const raw = request?.headers?.["x-codex-routing-hint"]; + if (typeof raw !== "string") return ""; + const match = /(?:^|;)\s*model=([^;]+)/.exec(raw); + return match ? match[1].trim() : ""; +} + export function websocketTarget(baseUrl, pathname, search = "") { const httpUrl = new URL(`${String(baseUrl).replace(/\/$/, "")}${upstreamPath(pathname)}${search}`); httpUrl.protocol = httpUrl.protocol === "https:" ? "wss:" : "ws:"; @@ -211,6 +224,13 @@ export function handleResponsesUpgrade({ rejectUpgrade(socket); return; } + const hintedModel = routingHintModel(request); + if (deepSeekModelFor(hintedModel)) { + // 426 triggers the client's direct FallbackToHttp path (no retries). + logger?.info?.(`deepseek websocket upgrade rejected with 426 ${pathname} (hint model=${hintedModel})`); + rejectUpgrade(socket); + return; + } wss.handleUpgrade(request, socket, head, (client) => { proxyChatGptSocket(client, { pathname, diff --git a/test/proxy.test.mjs b/test/proxy.test.mjs index 550e6a7..97a1640 100644 --- a/test/proxy.test.mjs +++ b/test/proxy.test.mjs @@ -6,7 +6,7 @@ import { once } from "node:events"; import { gzipSync, zstdCompressSync } from "node:zlib"; import { WebSocket, WebSocketServer } from "ws"; import { buildDeepSeekBody, createProxyServer } from "../src/proxy.mjs"; -import { requestModel, safeCloseCode, websocketTarget } from "../src/websocket-proxy.mjs"; +import { requestModel, routingHintModel, safeCloseCode, websocketTarget } from "../src/websocket-proxy.mjs"; const ROUTER_TOKEN = "A".repeat(43); @@ -1078,3 +1078,64 @@ test("upstream connect refusal closes the Codex websocket with 1011", async (t) ]); assert.equal(code, 1011); }); + +test("routingHintModel reads the model from the Codex routing hint header", () => { + assert.equal( + routingHintModel({ headers: { "x-codex-routing-hint": "model=deepseek/deepseek-flash;tier=priority" } }), + "deepseek/deepseek-flash", + ); + assert.equal( + routingHintModel({ headers: { "x-codex-routing-hint": "model=gpt-5.6-sol" } }), + "gpt-5.6-sol", + ); + assert.equal(routingHintModel({ headers: {} }), ""); + assert.equal(routingHintModel({ headers: { "x-codex-routing-hint": "tier=priority" } }), ""); + assert.equal(routingHintModel({ headers: { "x-codex-routing-hint": 42 } }), ""); + assert.equal(routingHintModel(undefined), ""); +}); + +test("DeepSeek-hinted upgrade is rejected with 426 before any upstream dial", async (t) => { + const upstream = await listenWsUpstream(); + const proxy = createProxyServer({ + chatGptBaseUrl: upstream.url, + logger: { info() {}, error() {} }, + routerToken: ROUTER_TOKEN, + }); + const proxyUrl = await listen(proxy); + t.after(async () => { await close(proxy); await close(upstream.server); }); + const { port } = new URL(proxyUrl); + const socket = net.connect(Number(port), "127.0.0.1"); + await once(socket, "connect"); + socket.write( + `GET /${ROUTER_TOKEN}/v1/responses HTTP/1.1\r\nHost: 127.0.0.1\r\n` + + "Connection: Upgrade\r\nUpgrade: websocket\r\n" + + "Sec-WebSocket-Key: dGhlIHNhbXBsZSBub25jZQ==\r\nSec-WebSocket-Version: 13\r\n" + + "x-codex-routing-hint: model=deepseek/deepseek-flash;tier=priority\r\n\r\n", + ); + const [chunk] = await once(socket, "data"); + assert.match(chunk.toString(), /^HTTP\/1\.1 426 /); + assert.equal(upstream.state.sockets.length, 0); + socket.destroy(); + const health = await fetch(`${proxyUrl}/${ROUTER_TOKEN}/health`); + assert.equal(health.status, 200); +}); + +test("GPT-hinted upgrade still proxies to chatgpt.com", async (t) => { + const upstream = await listenWsUpstream(); + const proxy = createProxyServer({ + chatGptBaseUrl: upstream.url, + logger: { info() {}, error() {} }, + routerToken: ROUTER_TOKEN, + }); + const proxyUrl = await listen(proxy); + t.after(async () => { + for (const socket of upstream.state.sockets) socket.terminate(); + await close(proxy); + await close(upstream.server); + }); + const { client, opened } = openClient(wsRoute(proxyUrl), { "x-codex-routing-hint": "model=gpt-5.6-sol" }); + t.after(() => { try { client.terminate(); } catch { /* already closed */ } }); + await opened; + await waitUntil(() => upstream.state.sockets.length >= 1); + assert.equal(upstream.state.sockets.length, 1); +});