From 4caac9be626ed03acae36dab6448a9332262d284 Mon Sep 17 00:00:00 2001 From: Raylan LIN Date: Thu, 1 Oct 2026 13:26:54 +0000 Subject: [PATCH] fix: P131 current-model request bodies, sidecar stall + P130 follow-ups; recommend GPT-6 Astra / Claude Opus 5.5 / Fable 5.1 Model support (every current Claude model and GPT-6 Astra were unusable): - Anthropic: stop sending temperature to Fable 5/5.1, Opus 4.7+, Sonnet 5+ (400 on every request); map reasoning level to adaptive thinking + output_config.effort instead of budget_tokens; never send thinking.disabled to always-thinking models; retry once without sampling/thinking/effort on a 400 naming them; surface stop_reason "refusal" instead of an empty reply. - OpenAI: treat gpt-6* as a reasoning model (max_completion_tokens, no temperature); omit reasoning_effort with tools on GPT-6 chat/completions and map 'minimal' to 'low'; recognise "... are not supported" as a reasoning-param error; Test connection and the vision path use max_completion_tokens for reasoning models. Sidecar: - The P129 stdin watchdog peeked stdin from a helper thread; after ~2 s idle, back-to-back requests stalled into timeouts. It now watches the parent PID. - Requests go out one at a time (budgets start when sent); heartbeats can extend a call to at most 3x its budget; onStillRunning fires on heartbeats; Stop cancels the wait (CANCELLED); failed results are cached under op_id so the TIMEOUT follow-up never re-runs a half-built generator; op_ids are namespaced per run. - Python < 3.11: catch concurrent.futures.TimeoutError in the heartbeat loop. - Durations reach the tool row and exports; the amber busy dot is wired up. - looksLikeQuestion judges the reply's ending (plans proceed, parameter requests stop). Settings / config: - Protocol switch and quick-fill apply a matching model and provider defaults. - Number fields clamp on blur; env fallback keeps saved preferences and is never persisted; no config save before the stored config has loaded. - Non-streaming requests no longer hit the 20 s connect timeout (and are not re-sent). - Truncation counts the system prompt + tool schemas; the max-rounds summary passes tools (Anthropic 400s on tool blocks without tools); theme/locale IPC channels moved into ipc-channels.ts; ruff E702 in test_reliability.py. Docs: presets, README / README.zh-CN, USER-GUIDE, ARCHITECTURE and .env.example recommend gpt-6-astra (OpenAI) and claude-opus-5-5 / claude-fable-5-1 (Anthropic). Tests: tests/llm-request.test.mjs (request bodies per model, truncation, llmFetch), tests/sidecar-client.test.mjs (Node client vs the real sidecar server; 5/5 fail on 0.2.129), Python tests for failure caching, futures timeout, PID watchdog. 191 JS + 58 Python tests pass; typecheck, lint, ruff, compileall, renderer build OK. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015nT6r6mw2MoBrJhEfWJbQD --- .env.example | 4 +- CHANGELOG.md | 97 ++++++++++ README.md | 16 +- README.zh-CN.md | 16 +- docs/ARCHITECTURE.md | 4 +- docs/ARCHITECTURE.zh-CN.md | 4 +- docs/USER-GUIDE.md | 4 +- docs/USER-GUIDE.zh-CN.md | 4 +- package.json | 4 +- sidecar/sw_agent/server.py | 71 +++++-- sidecar/tests/test_reliability.py | 62 +++++- src/main/agent/agent-loop-sidecar.ts | 50 ++++- src/main/com/sw-sidecar.ts | 97 ++++++++-- src/main/ipc/handlers.ts | 28 +-- src/main/llm/anthropic.ts | 77 +++++--- src/main/llm/context-window.ts | 8 +- src/main/llm/net.ts | 22 ++- src/main/llm/openai.ts | 35 +++- src/main/llm/thinking.ts | 131 ++++++++++++- src/main/llm/vision.ts | 27 ++- src/main/store/config.ts | 19 +- src/preload/index.ts | 8 +- src/renderer/App.tsx | 23 +-- src/renderer/components/SettingsModal.tsx | 77 ++++++-- src/renderer/components/Sidebar.tsx | 2 +- src/renderer/components/ToolCallGroup.tsx | 4 +- src/renderer/hooks/useLLM.ts | 4 +- src/renderer/session-export.ts | 4 +- src/shared/ipc-channels.ts | 6 + src/shared/presets.ts | 12 +- src/shared/types.ts | 3 + tests/agent-loop.test.mjs | 25 +++ tests/llm-request.test.mjs | 220 ++++++++++++++++++++++ tests/sidecar-client.test.mjs | 122 ++++++++++++ 34 files changed, 1128 insertions(+), 162 deletions(-) create mode 100644 tests/llm-request.test.mjs create mode 100644 tests/sidecar-client.test.mjs diff --git a/.env.example b/.env.example index 6291cdb..e83cdb3 100644 --- a/.env.example +++ b/.env.example @@ -7,12 +7,12 @@ # Anthropic # ANTHROPIC_API_KEY=sk-ant-xxxxx -# ANTHROPIC_MODEL=claude-sonnet-4-20250514 +# ANTHROPIC_MODEL=claude-opus-5-5 # OpenAI / 兼容服务 # OPENAI_API_KEY=sk-xxxxx # OPENAI_BASE_URL=https://api.openai.com/v1 -# OPENAI_MODEL=gpt-4o +# OPENAI_MODEL=gpt-6-astra # 百炼 # DASHSCOPE_API_KEY=sk-xxxxx diff --git a/CHANGELOG.md b/CHANGELOG.md index 1fbecbe..639368d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,103 @@ ## [Unreleased] +## [0.2.130] - 2026-10-01 + +### Fixed (P131 — current models, and what P129/P130 broke) + +**Every current Claude model was unusable.** The Anthropic adapter sent `temperature: 0.3` +on every request. Fable 5 / 5.1, Opus 4.7+ and Sonnet 5+ reject sampling parameters with a +400, and the retry only fired for errors that mentioned `thinking`. So every agent turn +failed on two of the three Claude presets. Reasoning levels were sent as `budget_tokens`, +which those models also reject (`off` sent `thinking.disabled`, a 400 on Opus 5.5). +- **Per-model request surface:** `thinking.ts` gains `claudeCaps()` / + `anthropicReasoningParams()`. These models now get no sampling fields, and reasoning + level maps to `thinking: {type:'adaptive', display:'summarized'}` + `output_config.effort`. + `off` on an always-thinking model becomes `effort: low`. +- **Broader retry:** a 400 naming temperature / thinking / effort is retried once without + those fields, so a model id we don't know yet degrades instead of dying. +- **Refusals:** `stop_reason: "refusal"` now shows a notice instead of an empty reply. + +**GPT-6 Astra was unusable on api.openai.com.** +- The reasoning-model check only matched `gpt-5`, so `gpt-6-astra` got `max_tokens` + + `temperature`, both 400s. +- On `/chat/completions` it also rejects `reasoning_effort` together with tools, and + rejects the `minimal` effort. Agent turns now omit the effort (model default), and plain + chat maps `off` → `low`. +- The fallback detector now recognises "… are not supported" errors. +- "Test connection" sent `max_tokens` to every OpenAI reasoning model, so it failed for + GPT-5.x too. It now uses `max_completion_tokens`. +- The vision path had the same `max_tokens` / `temperature` problem. + +**Sidecar stalls (since P129).** The stdin watchdog called `peek()` on stdin from a helper +thread. After ~2 s idle, the request that followed the next one was not read until more +input arrived, so back-to-back tool calls stalled into 60 s timeouts. That is likely the +real source of the "timed out but it finished" reports P130 worked around. The watchdog now +checks the parent PID and never touches stdin. + +**P130 timeout ≠ failure, finished properly.** +- **Absolute cap:** heartbeats prove Python is alive, not that SolidWorks progresses. A COM + call stuck on a modal dialog heartbeated forever, and Stop could not break it (every new + message got AGENT_BUSY). Each call now has an absolute cap of 3× its budget (`sw_status` + has none), and `call()` takes the abort signal (code `CANCELLED`). +- **One request at a time:** a request queued behind a slow call burned its budget unread, + timed out, and its same-op_id retry re-ran a tool whose first run had failed (only + successes were cached). Requests now go out one at a time, budgets start when sent, and + failures are cached under their op_id too. +- **Namespaced op_ids:** op_ids are scoped per agent run. Providers that restart tool-call + ids per conversation (Kimi's `functions.:`) could otherwise receive an older + session's cached result. +- **Python 3.9 / 3.10:** `concurrent.futures.TimeoutError` is not the builtin before 3.11, + so the first heartbeat tick escaped as an empty TOOL_FAILED. +- **"Still running" note:** `onStillRunning` only fired after a timeout that heartbeats + prevented, so the note never appeared. It now fires on heartbeats, once a minute in chat. +- **Durations:** tool durations were dropped before reaching the step. They now show on the + tool row and in exports. +- **Amber busy dot:** the busy state never reached `StatusDot`. It is wired through now and + serves `connected:true` while a tool runs; concurrent status probes share one request. +- **`looksLikeQuestion`:** it now judges the reply's ending. Plans mentioning + which / 几个 / 哪个 / 多少 no longer read as questions, and + 「我还缺少以下参数…」 / "I need a few values" no longer get auto-nudged. + +**Settings and config** +- **Protocol switch:** switching to the OpenAI protocol paired `api.openai.com` with + `deepseek-v4-pro`. It now uses that endpoint's suggested model. +- **Quick-fill buttons:** they now apply the provider's model, context window and max + output. Those defaults were declared in `presets.ts` but never used. +- **Number fields:** they clamp on blur, not per keystroke. Typing "2" used to snap to 4096, + so values like 200000 could not be entered. +- **Env-fallback config:** an env-sourced config no longer replaces the saved preferences, + and the env key is never persisted to the store. +- **Config load:** saving before the stored config loaded overwrote it with the defaults + and wiped the API key. +- **Non-streaming timeout:** the 20 s connect-stage timeout no longer acts as a total timeout + for non-streaming requests (vision captions), and no longer re-sends finished-but-slow + requests up to 3×. +- **Truncation:** agent-loop truncation now counts the system prompt and tool schemas, and + never returns a lone tool result. The max-rounds summary turn passes the tool list, since + Anthropic 400s on tool blocks without tools. +- **IPC channels:** theme / locale channels moved into `ipc-channels.ts`. +- **CI:** `sidecar/tests/test_reliability.py` failed ruff E702, so the next CI run would + have been red. + +### Changed +- **Recommended models:** GPT-6 Astra (`gpt-6-astra`, 1.05M context) for OpenAI; + Claude Opus 5.5 (`claude-opus-5-5`) and Claude Fable 5.1 (`claude-fable-5-1`) for + Anthropic. +- **Where they changed:** presets, provider defaults, README / README.zh-CN, USER-GUIDE, + ARCHITECTURE, `.env.example`. Saved model ids that left the preset list still load as + "Custom model". + +### Tests +- **`tests/llm-request.test.mjs`:** the request bodies each adapter sends, per model, plus + truncation and `llmFetch`. +- **`tests/sidecar-client.test.mjs`:** the Node client against the real sidecar server with + fake tools: queueing, failure caching, cap + progress, cancel, and the idle-gap stall. + It fails 5/5 on 0.2.129. +- **Python:** failure caching, the futures timeout on <3.11, and the PID watchdog. +- **`looksLikeQuestion`:** plan and question cases. +- **Totals:** 191 JS + 58 Python tests. + ## [0.2.129] - 2026-09-12 ### Changed (P130 — timeout≠failure) diff --git a/README.md b/README.md index 11ba339..d3b0b65 100644 --- a/README.md +++ b/README.md @@ -28,12 +28,12 @@

- version + version electron react typescript python - tests + tests license

@@ -86,7 +86,7 @@ Millwright: - **Agentic tool loop.** Observe → reason → act. The model chains multiple tool calls, reads structured JSON back from each one, and recovers from errors instead of failing silently. - **Visual understanding.** Reorient, rotate, screenshot, and analyze the model — via a multimodal main model or a dedicated vision model. - **Resident execution engine.** A persistent Python sidecar holds one COM connection open across an entire multi-step task. -- **Developer-friendly.** 167 TypeScript/Node tests plus a Python suite (`pytest sidecar/tests`) for the sidecar, a typed IPC boundary, and a `SKIP_SW_CONNECT` mode for UI-only development without SolidWorks installed. +- **Developer-friendly.** 191 TypeScript/Node tests plus a Python suite (`pytest sidecar/tests`) for the sidecar, a typed IPC boundary, and a `SKIP_SW_CONNECT` mode for UI-only development without SolidWorks installed. ## Cross-version compatibility @@ -160,17 +160,19 @@ A `Millwright-*-x64.zip` is also published alongside the Setup installer for use | Provider | Protocol | Base URL | Suggested model | |---|---|---|---| +| OpenAI | OpenAI | `https://api.openai.com/v1` | `gpt-6-astra` (GPT-6 Astra) | +| Anthropic | Anthropic | `https://api.anthropic.com` | `claude-opus-5-5` (Opus 5.5, default) / `claude-fable-5-1` (Fable 5.1, most capable) | | DeepSeek | OpenAI-compatible | `https://api.deepseek.com` | `deepseek-v4-pro` | | Kimi / Moonshot | OpenAI-compatible | `https://api.moonshot.cn/v1` | `kimi-k3` | | MiniMax | OpenAI-compatible | `https://api.minimaxi.com/v1` | `minimax-m3` | -| Anthropic | Anthropic | `https://api.anthropic.com` | `claude-sonnet-5` / `claude-opus-5` | -| OpenAI | OpenAI | `https://api.openai.com/v1` | `gpt-5.6` | | Alibaba Bailian (Qwen) | OpenAI-compatible | `https://dashscope.aliyuncs.com/compatible-mode/v1` | `qwen-3.8max` | | Zhipu (GLM) | OpenAI-compatible | `https://open.bigmodel.cn/api/paas/v4` | `glm-4.6` | | SiliconFlow | OpenAI-compatible | `https://api.siliconflow.cn/v1` | — | | Ollama (local) | OpenAI-compatible | `http://localhost:11434/v1` | — | -> Model IDs move fast — check your provider's docs for the current lineup. Agentic tool calling requires a model that supports function calling; DeepSeek V4, Kimi K3, MiniMax M3, and GLM-4.6 are first-class targets. +> Model IDs move fast — check your provider's docs for the current lineup. Agentic tool calling requires a model that supports function calling; GPT-6 Astra, Claude Opus 5.5 / Fable 5.1, DeepSeek V4, Kimi K3, MiniMax M3, and GLM-4.6 are first-class targets. +> +> The newest models reject request fields that older ones accepted: GPT-6 Astra needs `max_completion_tokens`, takes no `temperature`, and refuses `reasoning_effort` alongside tools on `/chat/completions`; Claude Opus 5.5 and Fable 5.1 reject `temperature` and `budget_tokens` and always think (depth is set with `effort`). Millwright detects these models by ID and sends the right fields — the **Reasoning depth** setting maps onto each model's own controls. ## Examples @@ -244,7 +246,7 @@ Contributions welcome — see [CONTRIBUTING.md](docs/CONTRIBUTING.md). We especi - [x] **v0.1** — MVP: Electron shell, LLM adapters, COM bridge, first tool set - [x] **v0.2** — Python sidecar, agentic tool loop, dual-engine fallback, vision feedback, confirmation cards, Apache-2.0 open source -- [x] **v0.2.4 → v0.2.37** — Extensive hardening against real SolidWorks installs ← *current*: the sketch → feature → cut → visual-verification loop now runs end to end on real hardware +- [x] **v0.2.4 → v0.2.130** — Extensive hardening against real SolidWorks installs ← *current*: the sketch → feature → cut → visual-verification loop now runs end to end on real hardware - [ ] **v0.3** — Streaming tool calls, sketching on model faces (not just reference planes), hole wizard, sheet metal, drawing annotations, remaining `#VERIFY` parameters confirmed - [ ] **v1.0** — MCP server, multi-CAD support diff --git a/README.zh-CN.md b/README.zh-CN.md index 9b4f79c..ddb3835 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -27,12 +27,12 @@

- version + version electron react typescript python - tests + tests license

@@ -85,7 +85,7 @@ Millwright: - **Agent 工具循环。** 观察 → 推理 → 执行。模型串联多次工具调用,读取每次返回的结构化 JSON,出错能自愈而不是静默失败。 - **视觉理解。** 可翻转、旋转、截屏,再做分析——既支持多模态主模型,也支持独立视觉模型。 - **常驻执行引擎。** 常驻 Python 边车在一整个多步任务中复用同一条 COM 连接。 -- **开发者友好。** 167 个 TS/Node 单元测试,另有独立的 Python 测试套件(`pytest sidecar/tests`),类型化 IPC 边界,`SKIP_SW_CONNECT` 纯 UI 开发模式(无需 SolidWorks)。 +- **开发者友好。** 191 个 TS/Node 单元测试,另有独立的 Python 测试套件(`pytest sidecar/tests`),类型化 IPC 边界,`SKIP_SW_CONNECT` 纯 UI 开发模式(无需 SolidWorks)。 ## 跨版本兼容 @@ -159,8 +159,8 @@ npm run dev | 服务商 | 协议 | Base URL | 推荐模型 | | ----------- | --------- | --------------------------------------------------- | ----------------------------------- | -| OpenAI | OpenAI | `https://api.openai.com/v1` | `gpt-5.6-sol` | -| Anthropic | Anthropic | `https://api.anthropic.com` | `claude-fable-5`(旗舰)/ `claude-opus-4-8`(更稳) | +| OpenAI | OpenAI | `https://api.openai.com/v1` | `gpt-6-astra`(GPT-6 Astra) | +| Anthropic | Anthropic | `https://api.anthropic.com` | `claude-opus-5-5`(Opus 5.5,默认)/ `claude-fable-5-1`(Fable 5.1,最强) | | DeepSeek | OpenAI 兼容 | `https://api.deepseek.com` | `deepseek-v4-pro`(强) / `deepseek-v4-flash`(快) | | Kimi / 月之暗面 | OpenAI 兼容 | `https://api.moonshot.cn/v1` | `kimi-k2.5` | | MiniMax | OpenAI 兼容 | `https://api.minimaxi.com/v1` | `minimax-m3`(512K 上下文) | @@ -169,7 +169,9 @@ npm run dev | 硅基流动 | OpenAI 兼容 | `https://api.siliconflow.cn/v1` | —(用户自填) | | Ollama(本地) | OpenAI 兼容 | `http://localhost:11434/v1` | —(用户自填) | -> 各家型号更新很快,请以服务商官方文档为准。Agent 工具调用需要模型支持 function calling;DeepSeek V4、Kimi K2、MiniMax M3、GLM-4.6 是一等公民。OpenAI `gpt-5.x` / o 系列和 Anthropic `claude-fable-5` 需要特定字段名,Millwright 已自动识别。 +> 各家型号更新很快,请以服务商官方文档为准。Agent 工具调用需要模型支持 function calling;GPT-6 Astra、Claude Opus 5.5 / Fable 5.1、DeepSeek V4、Kimi K2、MiniMax M3、GLM-4.6 是一等公民。 +> +> 新一代模型会拒绝老模型能接受的请求字段:GPT-6 Astra 要求 `max_completion_tokens`、不接受 `temperature`,并且在 `/chat/completions` 上不允许 `reasoning_effort` 与工具同时出现;Claude Opus 5.5 和 Fable 5.1 拒绝 `temperature` 和 `budget_tokens`,并且始终开启思考(深度用 `effort` 控制)。Millwright 按模型 ID 自动识别并发送正确的字段——设置里的「推理深度」会映射到各模型自己的控制参数上。OpenAI `gpt-5.x` / o 系列同样已自动识别。 ## 使用示例 @@ -243,7 +245,7 @@ SolidWorks - [x] **v0.1** — MVP:Electron 骨架、LLM 适配器、COM 桥接、首批工具 - [x] **v0.2** — Python 边车、agent 工具循环、双引擎降级、视觉反馈、确认卡片,Apache-2.0 开源 -- [x] **v0.2.4 → v0.2.37** — 大量真机加固 ← *当前*:草图 → 特征 → 切除 → 视觉核验的完整闭环已在真机上端到端跑通 +- [x] **v0.2.4 → v0.2.130** — 大量真机加固 ← *当前*:草图 → 特征 → 切除 → 视觉核验的完整闭环已在真机上端到端跑通 - [ ] **v0.3** — 流式工具调用、在模型面上画草图(而非仅基准面)、孔向导、钣金、工程图标注、剩余 `# VERIFY` 参数完成核验 - [ ] **v1.0** — MCP server、多 CAD 支持 diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index fa9d3b0..6262c57 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -213,8 +213,8 @@ const DEFAULT_SYSTEM_PROMPT = `You are a SolidWorks automation specialist. | Provider | Protocol | Base URL | Example model | |---------|----------|----------|----------| -| Anthropic | anthropic | https://api.anthropic.com | claude-sonnet-4-20250514 | -| OpenAI | openai | https://api.openai.com/v1 | gpt-4o | +| Anthropic | anthropic | https://api.anthropic.com | claude-opus-5-5 | +| OpenAI | openai | https://api.openai.com/v1 | gpt-6-astra | | Bailian | openai | https://dashscope.aliyuncs.com/compatible-mode/v1 | qwen-coder-plus | | MiniMax | openai | https://api.minimax.chat/v1 | MiniMax-Text-01 | | DeepSeek | openai | https://api.deepseek.com | deepseek-chat | diff --git a/docs/ARCHITECTURE.zh-CN.md b/docs/ARCHITECTURE.zh-CN.md index e10d82b..31d900c 100644 --- a/docs/ARCHITECTURE.zh-CN.md +++ b/docs/ARCHITECTURE.zh-CN.md @@ -213,8 +213,8 @@ const DEFAULT_SYSTEM_PROMPT = `你是一个 SolidWorks 自动化专家助手。 | 服务商 | 协议 | Base URL | 模型示例 | |--------|------|----------|----------| -| Anthropic | anthropic | https://api.anthropic.com | claude-sonnet-4-20250514 | -| OpenAI | openai | https://api.openai.com/v1 | gpt-4o | +| Anthropic | anthropic | https://api.anthropic.com | claude-opus-5-5 | +| OpenAI | openai | https://api.openai.com/v1 | gpt-6-astra | | 百炼 | openai | https://dashscope.aliyuncs.com/compatible-mode/v1 | qwen-coder-plus | | MiniMax | openai | https://api.minimax.chat/v1 | MiniMax-Text-01 | | DeepSeek | openai | https://api.deepseek.com | deepseek-chat | diff --git a/docs/USER-GUIDE.md b/docs/USER-GUIDE.md index 55f01f4..05e7a21 100644 --- a/docs/USER-GUIDE.md +++ b/docs/USER-GUIDE.md @@ -42,7 +42,7 @@ If you have an Anthropic API key: | API Protocol | Anthropic | | Base URL | https://api.anthropic.com | | API Key | Your `sk-ant-...` key | -| Model | `claude-sonnet-4-20250514` (recommended) | +| Model | `claude-opus-5-5` (recommended) or `claude-fable-5-1` (most capable) | Get your API key by signing up at console.anthropic.com and creating a key. @@ -53,7 +53,7 @@ Get your API key by signing up at console.anthropic.com and creating a key. | API Protocol | OpenAI-compatible | | Base URL | https://api.openai.com/v1 | | API Key | Your `sk-...` key | -| Model | `gpt-4o` | +| Model | `gpt-6-astra` (recommended) | ### Option 3: Bailian (Alibaba Cloud) diff --git a/docs/USER-GUIDE.zh-CN.md b/docs/USER-GUIDE.zh-CN.md index 6e1a82d..ec85495 100644 --- a/docs/USER-GUIDE.zh-CN.md +++ b/docs/USER-GUIDE.zh-CN.md @@ -42,7 +42,7 @@ | API 协议 | Anthropic | | Base URL | https://api.anthropic.com | | API Key | 你的 sk-ant-... 密钥 | -| 模型 | claude-sonnet-4-20250514(推荐) | +| 模型 | `claude-opus-5-5`(推荐)或 `claude-fable-5-1`(最强) | 获取 API Key:前往 console.anthropic.com 注册并创建密钥。 @@ -53,7 +53,7 @@ | API 协议 | OpenAI 兼容 | | Base URL | https://api.openai.com/v1 | | API Key | 你的 sk-... 密钥 | -| 模型 | gpt-4o | +| 模型 | `gpt-6-astra`(推荐) | ### 方式三:使用百炼(阿里云) diff --git a/package.json b/package.json index f4bc3c5..eab98c6 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "millwright", - "version": "0.2.129", + "version": "0.2.130", "description": "Open-source AI automation for SolidWorks — talk to your CAD.", "keywords": [ "solidworks", @@ -30,7 +30,7 @@ "pack": "npm run build && electron-builder --dir", "dist": "powershell -ExecutionPolicy Bypass -File scripts/prepare-python.ps1 && npm run build && electron-builder --publish never", "typecheck": "tsc -p tsconfig.main.json --noEmit && tsc -p tsconfig.preload.json --noEmit && tsc -p tsconfig.renderer.json --noEmit", - "test": "npm run build:main && node --test tests/agent-loop.test.mjs tests/code-extract.test.mjs tests/env-fallback.test.mjs tests/errors.test.mjs tests/factory.test.mjs tests/generators.test.mjs tests/presets.test.mjs tests/sanitizer.test.mjs tests/sse.test.mjs tests/sw-tools.test.mjs tests/vba-helpers.test.mjs tests/vba-macro-writer.test.mjs tests/housekeeping.test.mjs", + "test": "npm run build:main && node --test tests/agent-loop.test.mjs tests/code-extract.test.mjs tests/env-fallback.test.mjs tests/errors.test.mjs tests/factory.test.mjs tests/generators.test.mjs tests/presets.test.mjs tests/sanitizer.test.mjs tests/sse.test.mjs tests/sw-tools.test.mjs tests/vba-helpers.test.mjs tests/vba-macro-writer.test.mjs tests/housekeeping.test.mjs tests/llm-request.test.mjs tests/sidecar-client.test.mjs", "lint": "eslint src/ --ext .ts,.tsx", "format": "prettier --write src/", "clean": "rimraf dist release" diff --git a/sidecar/sw_agent/server.py b/sidecar/sw_agent/server.py index 6c73478..0bff7fd 100644 --- a/sidecar/sw_agent/server.py +++ b/sidecar/sw_agent/server.py @@ -28,6 +28,7 @@ import threading import time from collections import OrderedDict +from concurrent.futures import TimeoutError as FutureTimeout from sw_agent import registry, session_log, verify from sw_agent.bridge import Context, SWError, hresult, is_dead_connection @@ -97,6 +98,18 @@ def bump(self) -> int: GUARD = _Guard() +class _Failed: + """P131: a failed call, cached under its op_id like a result. A retry with the same + op_id (Node's TIMEOUT follow-up) must get the failure back — before P131 only + successes were cached, so the retry RE-RAN a generator that had built half a gear + and then raised.""" + + __slots__ = ("exc",) + + def __init__(self, exc: Exception) -> None: + self.exc = exc + + class CodedError(Exception): def __init__(self, code: str, message: str) -> None: super().__init__(message) @@ -164,6 +177,8 @@ def _advisory(ctx: Context) -> dict | None: def _call(ctx: Context, name: str, args: dict, op_id=None, expect_state=None): """One tool call on the COM thread. Dead connection → reconnect once and retry.""" cached = GUARD.get(op_id) + if isinstance(cached, _Failed): + raise cached.exc if cached is not None: dup = dict(cached) if isinstance(cached, dict) else {"result": cached} dup["_duplicate"] = True @@ -205,9 +220,14 @@ def work(): return work() except Exception as e: if not is_dead_connection(e): + GUARD.put(op_id, _Failed(e)) raise - ctx.reconnect() + ctx.reconnect() + try: return work() + except Exception as e: + GUARD.put(op_id, _Failed(e)) + raise HEARTBEAT_S = 15.0 @@ -222,7 +242,10 @@ def _call_with_heartbeat(executor, ctx, rid, name, args, op_id, expect): while True: try: return fut.result(timeout=HEARTBEAT_S) - except TimeoutError: + except (TimeoutError, FutureTimeout): + # P131: before Python 3.11 concurrent.futures.TimeoutError is NOT the builtin + # TimeoutError — on 3.9/3.10 the first tick escaped as an empty TOOL_FAILED + # while the tool kept running (and the agent retried it). _write({"id": rid, "progress": True, "tool": name, "elapsed_ms": round((time.perf_counter() - t0) * 1000)}) @@ -246,6 +269,27 @@ def _health(ctx: Context) -> dict: return info +def _parent_alive(ppid: int) -> bool: + """P131: is the process that spawned us still running? Checked by PID, never via stdin.""" + if os.name == "nt": + import ctypes + from ctypes import wintypes + k32 = ctypes.WinDLL("kernel32", use_last_error=True) + k32.OpenProcess.argtypes = (wintypes.DWORD, wintypes.BOOL, wintypes.DWORD) + k32.OpenProcess.restype = wintypes.HANDLE + k32.WaitForSingleObject.argtypes = (wintypes.HANDLE, wintypes.DWORD) + k32.WaitForSingleObject.restype = wintypes.DWORD + k32.CloseHandle.argtypes = (wintypes.HANDLE,) + h = k32.OpenProcess(0x00100000, False, ppid) # SYNCHRONIZE + if not h: + return ctypes.get_last_error() != 87 # ERROR_INVALID_PARAMETER: no such process + try: + return k32.WaitForSingleObject(h, 0) == 0x102 # WAIT_TIMEOUT: still running + finally: + k32.CloseHandle(h) + return os.getppid() == ppid # POSIX: an orphan is re-parented + + def serve() -> None: ctx = Context() executor = ComExecutor("sw-com") @@ -256,22 +300,23 @@ def serve() -> None: # blocks while SolidWorks loads add-ins sits at the HEAD of the queue and delays every # probe behind it — that is how the first sw_status of the 16:15 session timed out. # Connecting lazily costs a few hundred ms on the first real call and blocks nothing. + # If the parent dies mid-COM-call the stdin loop never sees EOF. A helper thread watches + # the parent PID: once it is gone, give the current job 5 s and hard-exit. + # P131: the P129 version peeked stdin from this thread. Two readers on one pipe — after + # ~2 s idle the peek parked inside the buffered reader, and the request that followed the + # next one was not read until more input arrived: back-to-back tool calls stalled into + # 60 s timeouts. The watchdog must never touch stdin. + ppid = os.getppid() + def _watchdog(): - # If the parent dies mid-COM-call the stdin loop never sees EOF. Poll the pipe from a - # helper thread: once stdin is closed, give the current job 5 s and hard-exit. try: - while not sys.stdin.closed: + while _parent_alive(ppid): time.sleep(2) - try: - if sys.stdin.buffer.peek(1) == b"": - break - except Exception: # noqa: BLE001 — peek on a closed stdin pipe is the trigger we care about - break - except Exception: # noqa: BLE001 — watchdog must never crash the sidecar itself - pass + except Exception: # noqa: BLE001 — a watchdog bug must never kill a healthy sidecar + return time.sleep(5) os._exit(0) - threading.Thread(target=_watchdog, name="stdin-watchdog", daemon=True).start() + threading.Thread(target=_watchdog, name="parent-watchdog", daemon=True).start() for raw in sys.stdin: raw = raw.strip() if not raw: diff --git a/sidecar/tests/test_reliability.py b/sidecar/tests/test_reliability.py index 8eeebcd..f81c63e 100644 --- a/sidecar/tests/test_reliability.py +++ b/sidecar/tests/test_reliability.py @@ -15,7 +15,9 @@ def test_guard_idempotency_and_bump(): g = server._Guard(cap=2) assert g.get(None) is None and g.get("x") is None - g.put("a", {"r": 1}); g.put("b", {"r": 2}); g.put("c", {"r": 3}) + g.put("a", {"r": 1}) + g.put("b", {"r": 2}) + g.put("c", {"r": 3}) assert g.get("a") is None and g.get("c") == {"r": 3} # LRU cap assert g.bump() == 1 and g.bump() == 2 @@ -60,3 +62,61 @@ class Ctx: # never reached: STALE_STATE is checked before any COM access server._call(Ctx(), "extrude", {}, op_id=None, expect_state=3) assert ei.value.code == "STALE_STATE" server.GUARD.state_version = 0 + + +def test_failed_call_is_cached_under_op_id(): + """P131: a TIMEOUT follow-up with the same op_id must get the failure back, not re-run.""" + from sw_agent import registry + + runs = [] + + def flaky(ctx): + runs.append(1) + raise SWError("gear body built, then the hole failed") + + registry.TOOLS["_p131_flaky"] = registry.ToolSpec("_p131_flaky", "", {}, "", False, True, flaky) + try: + for _ in range(2): + with pytest.raises(SWError): + server._call(object(), "_p131_flaky", {}, op_id="p131-op", expect_state=None) + assert len(runs) == 1 + finally: + registry.TOOLS.pop("_p131_flaky", None) + server.GUARD.done.pop("p131-op", None) + + +def test_heartbeat_survives_futures_timeout(monkeypatch): + """P131: on Python < 3.11 Future.result(timeout) raises concurrent.futures.TimeoutError, + which is not the builtin — the first heartbeat tick used to escape as TOOL_FAILED.""" + from concurrent.futures import Future + from concurrent.futures import TimeoutError as FutureTimeout + + frames = [] + monkeypatch.setattr(server, "_write", frames.append) + + class SlowFuture(Future): + ticks = 0 + + def result(self, timeout=None): + SlowFuture.ticks += 1 + if SlowFuture.ticks < 3: + raise FutureTimeout() # what 3.9 / 3.10 raise + return {"ok": True} + + class Exec: + def submit(self, fn): + return SlowFuture() + + assert server._call_with_heartbeat(Exec(), None, 7, "slow", {}, None, None) == {"ok": True} + assert [f["progress"] for f in frames] == [True, True] + assert all(f["id"] == 7 for f in frames) + + +def test_parent_watchdog_uses_pid_not_stdin(): + """P131: the P129 watchdog peeked stdin from a second thread and stalled requests.""" + import inspect + + assert server._parent_alive(os.getppid()) is True + if os.name != "nt": + assert server._parent_alive(os.getppid() + 999_999) is False + assert "peek(" not in inspect.getsource(server.serve) diff --git a/src/main/agent/agent-loop-sidecar.ts b/src/main/agent/agent-loop-sidecar.ts index b3a0996..96d081b 100644 --- a/src/main/agent/agent-loop-sidecar.ts +++ b/src/main/agent/agent-loop-sidecar.ts @@ -50,6 +50,9 @@ export interface AgentEvent { export interface SidecarAgentOptions { requestId?: string; + /** P131: model id + the system prompt the adapter sends — truncation budgets for them */ + model?: string; + systemPrompt?: string; maxRounds?: number; signal?: AbortSignal; onEvent?: (ev: AgentEvent) => void; @@ -189,6 +192,10 @@ export async function runSidecarAgent( let backupDone = false; // P125: surface SolidWorks version advisory once per session let advisoryShown = false; + // P131: op_ids are namespaced per run. The sidecar's idempotency cache outlives a chat + // session, and some providers restart their tool-call ids every conversation (Kimi's + // `functions.:`) — a bare call.id could hand a NEW call an OLD session's result. + const runNonce = Math.random().toString(36).slice(2, 10); // P19: cache the most recent screenshot so a (pure-text or multimodal) model can // ask several follow-up questions about the SAME snapshot without re-capturing. @@ -236,6 +243,8 @@ export async function runSidecarAgent( ...VIRTUAL_TOOLS.filter((t) => t.function.name !== 'run_macro' || !!opts.runMacro), ...sidecarTools, ].filter((t: any) => !off.has(t?.function?.name)); + // P131: text that rides along with every request besides the history + const overhead = (opts.systemPrompt ?? '') + JSON.stringify(tools); // P115: every advertised tool name — the audit scans narration for these. const toolNames = new Set(tools.map((t: any) => t?.function?.name).filter(Boolean)); const destructive = new Set( @@ -291,7 +300,9 @@ export async function runSidecarAgent( for (let round = 0; round < maxRounds; round++) { if (opts.signal?.aborted) throw new Error('已取消'); - history = truncateMessages(history, '', opts.requestId ? String(opts.requestId) : '', opts.contextWindow); + // P131: the request also carries the system prompt and every tool schema — count them + // (they never were: the prompt slot got '' and the model slot got the request id). + history = truncateMessages(history, overhead, opts.model ?? '', opts.contextWindow); // Thin guard (block-aware truncation should already prevent this) while (history.length > 0 && (history[0].role === 'tool' || (history[0].role === 'system' && history[0].toolCalls?.length))) { history.shift(); @@ -484,13 +495,21 @@ export async function runSidecarAgent( // P130: onStillRunning callback fires every time the sidecar sends a progress // heartbeat; we surface a "still running" note in the chat so the user knows the // tool is genuinely progressing instead of just hanging. + // P131: one note per minute of waiting (heartbeats arrive every 15 s), and Stop + // abandons the wait instead of leaving the run stuck until the app restarts. + let lastNote = 0; const r = await sidecar.call(call.name, call.parameters, { - opId: call.id, - onStillRunning: (ms) => opts.onEvent?.({ type: 'text', text: `(${call.name} 已运行 ${Math.round(ms / 1000)} s,SolidWorks 仍在执行,继续等待…)` }), + opId: `${runNonce}:${call.id}`, + signal: opts.signal, + onStillRunning: (ms) => { + if (ms - lastNote < 60_000) return; + lastNote = ms; + opts.onEvent?.({ type: 'text', text: `\n(${call.name} 已运行 ${Math.round(ms / 1000)} s,SolidWorks 仍在执行,继续等待…)\n` }); + }, }); // P130: stash the wall-clock duration on the call so the UI card and the session // export can show it. - (call as any).durationMs = r.durationMs; + call.durationMs = r.durationMs; // P125: surface unverified SolidWorks version advisory once if (r.ok && r.data?._advisory && !advisoryShown) { advisoryShown = true; @@ -517,7 +536,7 @@ export async function runSidecarAgent( // from a path not covered by the per-call guard above), the API rejects the next // request with "assistant message with 'tool_calls' must be followed by tool // messages". Scan backwards from the tail and answer any orphaned call_ids. - history = truncateMessages(history, '', '', opts.contextWindow); + history = truncateMessages(history, overhead, opts.model ?? '', opts.contextWindow); for (let i = history.length - 1; i >= 0; i--) { const m = history[i]; if (m.role !== 'assistant' || !m.toolCalls?.length) continue; @@ -536,7 +555,10 @@ export async function runSidecarAgent( role: 'system', content: '(已达到最大工具调用轮数。请不要再调用工具:总结你已完成的操作、当前模型的状态、以及未完成的部分。)', }); - const summary = await adapter.chatWithTools(history, opts.signal, undefined); + // P131: pass the real tool list — Anthropic rejects a history containing tool_use / + // tool_result blocks when the request declares no tools, so on every Claude model + // this summary 400'd and was silently lost. The system note above forbids calls. + const summary = await adapter.chatWithTools(history, opts.signal, tools); if (summary.content) { finalText = summary.content; opts.onEvent?.({ type: 'text', text: summary.content }); @@ -775,8 +797,20 @@ function auditClaimedCalls( * auto-nudged, and built a gear from the example values the user never confirmed. */ export function looksLikeQuestion(text: string): boolean { const t = text.trim(); - if (/[??]\s*$/.test(t)) return true; // ends with a question mark + if (!t) return false; + if (/[??]\s*[))"”'」]*\s*$/.test(t)) return true; // ends with a question mark if ((t.match(/[??]/g) || []).length >= 2) return true; // several questions inside - return /(请(告诉|提供|给出|确认|回复|选择)|需要(你|您)?(提供|确认|告诉)|告诉我|你想|您想|哪(一)?(种|个)|多少|几个|which|what (size|module|value|diameter)|please (provide|tell|confirm|specify)|let me know|do you want|would you like|could you (tell|confirm))/i.test(t); + // P131: judge the ENDING. The P130 pattern matched generic words anywhere (which / 多少 / + // 几个 / 哪个 / 你想), so ordinary plans read as questions and the nudge never fired; and + // "我还缺少以下参数…" / "I need a few values from you" were missed. + const tail = t.slice(-300); + if (PLAN_CLOSE.test(tail)) return false; + return ASKS_USER.test(tail); } +/** "…Starting now." / "…现在开始执行。" — the reply commits to acting. */ +const PLAN_CLOSE = /(现在开始|开始执行|开始建模|马上开始|立即开始|接下来(我)?(会|将)?(调用|执行|开始)|starting now|let me (start|begin)|i(?:'ll| will) (?:now )?(?:start|begin|proceed)|proceeding now)[^。.!!\n]*[。.!!]?\s*$/i; + +/** The reply hands the turn back: it asks for input the user has not given. */ +const ASKS_USER = /(请(告诉|提供|给出|确认|回复|选择|补充|说明)|需要(你|您)?(提供|确认|告诉|补充)|告诉我|还?缺少?(以下|这些|下列)?(参数|尺寸|信息|数值)|需要以下|请问|(你|您)希望|是否需要|please (provide|tell|confirm|specify|let me know)|let me know|i need (a few|some|the following|these|you to|more)|(could|can) you (tell|confirm|provide|specify)|do you want|would you like|what (size|module|value|diameter|dimensions?) (do|would|should))/i; + diff --git a/src/main/com/sw-sidecar.ts b/src/main/com/sw-sidecar.ts index dbb0f24..020a937 100644 --- a/src/main/com/sw-sidecar.ts +++ b/src/main/com/sw-sidecar.ts @@ -16,6 +16,16 @@ // sidecar's idempotency cache hands back the true result when the job finishes // - `inFlight` + `lastStatus` let the UI probe skip the queue while a tool is running // - every rpc logs its duration so the next report carries numbers, not guesses +// +// P131 — what P130 left broken: +// - heartbeats prove PYTHON is alive, not that SolidWorks progresses: a COM call blocked +// on a modal dialog heartbeated forever and nothing timed out. Each call now also has +// an absolute cap (3× its budget; sw_status none — the probe must fail fast). +// - the sidecar reads stdin one request at a time, so a request queued behind a slow +// call burned its budget before it was even read, timed out, and its retry re-ran the +// tool. Requests are now sent one at a time; a budget starts when its request is sent. +// - `onStillRunning` fired only after a timeout that heartbeats prevented — it now fires +// on every heartbeat. A `signal` lets Stop abandon the wait (code CANCELLED). import { spawn, ChildProcessWithoutNullStreams } from 'child_process'; import { randomUUID } from 'crypto'; @@ -27,7 +37,7 @@ import { resolvePythonPath, resolveSidecarCwd } from '../python-path'; export type SidecarErrorCode = | 'NO_CONNECTION' | 'NO_DOCUMENT' | 'WRONG_DOC_TYPE' | 'UNKNOWN_TOOL' - | 'BAD_ARGS' | 'STALE_STATE' | 'COM_ERROR' | 'TOOL_FAILED' | 'TIMEOUT'; + | 'BAD_ARGS' | 'STALE_STATE' | 'COM_ERROR' | 'TOOL_FAILED' | 'TIMEOUT' | 'CANCELLED'; export interface SidecarResult { ok: boolean; @@ -54,18 +64,32 @@ export interface CallOptions { expectState?: number; /** Idle budget for this call (ms). Defaults: SLOW_TOOLS table, then callTimeoutMs. */ timeoutMs?: number; - /** Fired when the first budget expires and we keep waiting on the same op_id. */ + /** Fired on every sidecar heartbeat while the call is still running. */ onStillRunning?: (elapsedMs: number) => void; + /** P131: abort → resolve CANCELLED at once. The sidecar finishes the job regardless and + * caches the outcome under the op_id. */ + signal?: AbortSignal; +} + +interface RpcOptions { + /** Idle budget (ms) — heartbeats reset it. */ + budget?: number; + /** P131: absolute deadline (ms) that heartbeats do NOT extend. Defaults to `budget`. */ + hardCapMs?: number; + onProgress?: (elapsedMs: number) => void; + signal?: AbortSignal; } interface Pending { resolve: (v: SidecarResult) => void; reject: (e: Error) => void; timer: ReturnType; + hardTimer?: ReturnType; budget: number; startedAt: number; label: string; onTimeout: () => void; + onProgress?: (elapsedMs: number) => void; } interface ReadyWaiter { resolve: () => void; reject: (e: Error) => void } @@ -108,6 +132,8 @@ export class SWSidecar { lastStatus: any = null; /** P130: name of the tool currently executing (first in-flight `call`), for the UI */ runningTool: string | null = null; + /** P131: requests go out one at a time — the sidecar reads stdin serially anyway */ + private tail: Promise = Promise.resolve(); constructor(opts: SidecarOptions = {}) { this.opts = { @@ -187,11 +213,14 @@ export class SWSidecar { if (msg.progress) { clearTimeout(p.timer); p.timer = setTimeout(p.onTimeout, p.budget); - this.opts.onLog?.(`[sidecar] ${p.label} still running (${Math.round((msg.elapsed_ms ?? 0) / 1000)}s)`); + const elapsed = Date.now() - p.startedAt; + this.opts.onLog?.(`[sidecar] ${p.label} still running (${Math.round(elapsed / 1000)}s)`); + try { p.onProgress?.(elapsed); } catch { /* a UI callback must never break the RPC */ } return; } this.pending.delete(msg.id); clearTimeout(p.timer); + clearTimeout(p.hardTimer); this.updateRunning(); const durationMs = Date.now() - p.startedAt; this.opts.onLog?.(`[sidecar] ${p.label} ${durationMs}ms ok=${!!msg.ok}${msg.code ? ' code=' + msg.code : ''}`); @@ -199,27 +228,53 @@ export class SWSidecar { } private updateRunning(): void { - const first = [...this.pending.values()].find((p) => p.label.startsWith('call:')); + // sw_status is the UI's own probe — never report it as "a tool is running" + const first = [...this.pending.values()].find((p) => p.label.startsWith('call:') && p.label !== 'call:sw_status'); this.runningTool = first ? first.label.slice(5) : null; } - private rpc(method: string, params?: any, budget?: number): Promise { + /** P131: queue behind whatever is outstanding; resolve CANCELLED as soon as `signal` + * aborts (a request not yet sent is then never sent). */ + private rpc(method: string, params?: any, o: RpcOptions = {}): Promise { + return new Promise((resolve) => { + let settled = false; + const settle = (r: SidecarResult) => { + if (settled) return; + settled = true; + o.signal?.removeEventListener('abort', onAbort); + resolve(r); + }; + const onAbort = () => settle({ ok: false, code: 'CANCELLED', error: '已取消' }); + if (o.signal?.aborted) return onAbort(); + o.signal?.addEventListener('abort', onAbort, { once: true }); + const turn = this.tail.then(() => (settled ? undefined : this.send(method, params, o).then(settle))); + this.tail = turn.catch(() => undefined); + }); + } + + private send(method: string, params: any, o: RpcOptions): Promise { if (!this.proc || !this.proc.stdin.writable) { return Promise.resolve({ ok: false, code: 'NO_CONNECTION', error: 'Python 组件未运行——请安装 Python + pywin32,或忽略此错误(将自动使用内置 VBS 引擎)' }); } const id = this.nextId++; const label = method === 'call' ? `call:${params?.name ?? '?'}` : method; - const b = budget ?? this.opts.callTimeoutMs; + const b = o.budget ?? this.opts.callTimeoutMs; + const cap = Math.max(o.hardCapMs ?? b, b); return new Promise((resolve, reject) => { const startedAt = Date.now(); const onTimeout = () => { + const p = this.pending.get(id); + if (!p) return; + clearTimeout(p.timer); + clearTimeout(p.hardTimer); this.pending.delete(id); this.updateRunning(); - this.opts.onLog?.(`[sidecar] ${label} TIMEOUT after ${Date.now() - startedAt}ms (budget ${b}ms, pending=${this.pending.size}, running=${this.runningTool ?? '-'})`); + this.opts.onLog?.(`[sidecar] ${label} TIMEOUT after ${Date.now() - startedAt}ms (budget ${b}ms, cap ${cap}ms, pending=${this.pending.size}, running=${this.runningTool ?? '-'})`); resolve({ ok: false, code: 'TIMEOUT', error: `Python 组件调用超时:${label}`, durationMs: Date.now() - startedAt }); }; const timer = setTimeout(onTimeout, b); - this.pending.set(id, { resolve, reject, timer, budget: b, startedAt, label, onTimeout }); + const hardTimer = cap > b ? setTimeout(onTimeout, cap) : undefined; + this.pending.set(id, { resolve, reject, timer, hardTimer, budget: b, startedAt, label, onTimeout, onProgress: o.onProgress }); this.updateRunning(); this.proc!.stdin.write(JSON.stringify({ id, method, params: params ?? {} }) + '\n'); }); @@ -233,7 +288,8 @@ export class SWSidecar { } /** Invoke a tool. A timeout is retried once with the SAME op_id (P125 idempotency cache): - * if the sidecar finished the job meanwhile, we get the real result, not a re-run. */ + * if the sidecar finished the job meanwhile, we get the real result, not a re-run + * (P131: failures are cached too, so a half-built generator is never re-run). */ async call(name: string, args?: Record, opts?: CallOptions): Promise { const opId = opts?.opId ?? randomUUID(); const params: any = { name, args: args ?? {} }; @@ -241,25 +297,33 @@ export class SWSidecar { if (opts?.expectState != null) params.expect_state = opts.expectState; const budget = opts?.timeoutMs ?? SLOW_TOOLS[name] ?? this.opts.callTimeoutMs; const t0 = Date.now(); - let r = await this.rpc('call', params, budget); + const o: RpcOptions = { + budget, + // P131: heartbeats may stretch a call to 3× its budget, never further; the probe not at all + hardCapMs: name === 'sw_status' ? budget : budget * 3, + onProgress: opts?.onStillRunning ? () => opts.onStillRunning!(Date.now() - t0) : undefined, + signal: opts?.signal, + }; + let r = await this.rpc('call', params, o); if (!r.ok && r.code === 'TIMEOUT' && name !== 'sw_status') { opts?.onStillRunning?.(Date.now() - t0); - r = await this.rpc('call', params, budget); // same op_id → cached result when the job lands + // same op_id → cached result when the job lands; one more budget, no extension + r = await this.rpc('call', params, { ...o, hardCapMs: budget }); if (!r.ok && r.code === 'TIMEOUT') { r.error = `工具 ${name} 执行超过 ${Math.round((Date.now() - t0) / 1000)} s 仍未返回。SolidWorks 可能在重建、加载插件或弹出了对话框;` + `操作可能已经完成——请先用 list_features 核对,不要重复执行。`; } } if (name === 'sw_status' && r.ok) this.lastStatus = r.data; - if (r.durationMs == null) r.durationMs = Date.now() - t0; + r.durationMs = Date.now() - t0; // P131: the whole wait, including a follow-up return r; } - ping(): Promise { return this.rpc('ping', undefined, 10_000); } - reconnect(): Promise { return this.rpc('reconnect', undefined, 30_000); } + ping(): Promise { return this.rpc('ping', undefined, { budget: 10_000 }); } + reconnect(): Promise { return this.rpc('reconnect', undefined, { budget: 30_000 }); } async health(): Promise> { - const r = await this.rpc('health', undefined, 10_000); + const r = await this.rpc('health', undefined, { budget: 10_000 }); if (!r.ok && /unknown method/.test(r.error ?? '')) { const p = await this.ping(); return { ok: p.ok, data: { connected: p.ok, state_version: 0, tool_count: 0 } as SidecarHealth, error: p.error }; @@ -268,7 +332,7 @@ export class SWSidecar { } async stateVersion(): Promise { - const r = await this.rpc('state', undefined, 10_000); + const r = await this.rpc('state', undefined, { budget: 10_000 }); return r.ok ? Number(r.data?.state_version ?? 0) : 0; } @@ -286,6 +350,7 @@ export class SWSidecar { this.rl = null; for (const [, p] of this.pending) { clearTimeout(p.timer); + clearTimeout(p.hardTimer); p.resolve({ ok: false, code: 'NO_CONNECTION', error: err.message }); } this.pending.clear(); diff --git a/src/main/ipc/handlers.ts b/src/main/ipc/handlers.ts index 11dfb1e..031b5d8 100644 --- a/src/main/ipc/handlers.ts +++ b/src/main/ipc/handlers.ts @@ -13,7 +13,7 @@ import { createAdapter, validateConfig } from '../llm'; import { truncateMessages } from '../llm/context-window'; import { resolveSystemPrompt } from '../llm/prompts'; import { getBridge } from '../com/sw-bridge'; -import { getSidecar } from '../com/sw-sidecar'; +import { getSidecar, type SidecarResult } from '../com/sw-sidecar'; import { collectDocumentContext, formatContextForPromptAsync, invalidateContextCache } from '../com/context-collector'; import { ScriptEngine } from '../scripts/engine'; import { validateScript } from '../scripts/sanitizer'; @@ -120,6 +120,7 @@ export function registerIpcHandlers(getMainWindow: () => BrowserWindow | null) { return { ok: true }; }); + let statusProbe: Promise | null = null; ipcMain.handle(IpcChannels.SW_STATUS, async () => { // P73: the sidecar holds the connection the tools actually run through. If it can read // ActiveDoc we ARE connected, whatever the separate cscript probe concludes — and it was @@ -130,10 +131,16 @@ export function registerIpcHandlers(getMainWindow: () => BrowserWindow | null) { // P130: a tool is executing on the sidecar's single COM thread — do not queue behind it. // Serve the last known status and say we are busy; the UI shows "working" instead of a // red/green flicker, and the probe can no longer time out in the queue. - if (sidecar.runningTool && sidecar.lastStatus) { - return { ...sidecar.lastStatus, source: 'sidecar' as const, busy: true, runningTool: sidecar.runningTool }; + // P131: also when no status was cached yet (a probe would only queue behind the tool), + // and always connected — a tool is running through this very connection, whatever + // an older probe said. + if (sidecar.runningTool) { + return { ...(sidecar.lastStatus ?? {}), connected: true, source: 'sidecar' as const, busy: true, runningTool: sidecar.runningTool }; } - const r = await sidecar.call('sw_status', {}); + // P131: the UI polls every 3 s; while SolidWorks is slow to answer, share the probe in + // flight instead of queueing another one behind it each tick. + statusProbe ??= sidecar.call('sw_status', {}).finally(() => { statusProbe = null; }); + const r = await statusProbe; if (r.ok && r.data?.connected) { return { ...r.data, source: 'sidecar' as const }; } @@ -317,6 +324,8 @@ export function registerIpcHandlers(getMainWindow: () => BrowserWindow | null) { if (sidecarReady) { const text = await runSidecarAgent(adapter, payload.messages, sidecar, { requestId, + model: enrichedConfig.model, + systemPrompt: enrichedConfig.systemPrompt, maxRounds: payload.config.maxRounds ?? 24, approvalMode: payload.config.approvalMode ?? 'normal', // P115: strong-prompt mode — L3 re-injects tool rules every round + @@ -477,23 +486,20 @@ export function registerIpcHandlers(getMainWindow: () => BrowserWindow | null) { }); // ===== Theme (kept independent of the LLM config) ===== - // Reusing the CONFIG_ channel-name convention felt too cramped, so we register - // two dedicated handlers here. The channel names piggyback on `config:save/load` - // for now; we can split them out later if needed. - ipcMain.handle('theme:load', async (): Promise => { + ipcMain.handle(IpcChannels.THEME_LOAD, async (): Promise => { return await loadTheme(); }); - ipcMain.handle('theme:save', async (_e, theme: ThemeName) => { + ipcMain.handle(IpcChannels.THEME_SAVE, async (_e, theme: ThemeName) => { await saveTheme(theme); return { ok: true }; }); - ipcMain.handle('locale:load', async (): Promise => { + ipcMain.handle(IpcChannels.LOCALE_LOAD, async (): Promise => { return await loadLocale(); }); - ipcMain.handle('locale:save', async (_e, locale: LocaleName) => { + ipcMain.handle(IpcChannels.LOCALE_SAVE, async (_e, locale: LocaleName) => { await saveLocale(locale); return { ok: true }; }); diff --git a/src/main/llm/anthropic.ts b/src/main/llm/anthropic.ts index 322fe3f..f15463e 100644 --- a/src/main/llm/anthropic.ts +++ b/src/main/llm/anthropic.ts @@ -23,7 +23,7 @@ import { resolveSystemPrompt } from './prompts'; import { extractFirstCodeBlock } from './code-extract'; import { LLMHttpError, extractErrorMessage, toLLMError } from './errors'; import { parseSSE } from './sse'; -import { splitThinking } from './thinking'; +import { splitThinking, anthropicReasoningParams } from './thinking'; import type { ChatMessage, LLMResponse, @@ -45,6 +45,7 @@ interface AnthropicResponseBody { content: AnthropicTextContent[]; model: string; stop_reason: string | null; + stop_details?: { category?: string | null } | null; usage?: { input_tokens: number; output_tokens: number; @@ -67,19 +68,14 @@ export class AnthropicAdapter extends BaseLLMAdapter { } /** - * P53: extended thinking. Anthropic takes a budget, not a level, and requires - * budget_tokens < max_tokens. 'auto' sends nothing (model default). + * P53: reasoning depth. P131: per-model request surface (see anthropicReasoningParams) + * — current Claude models reject `temperature` and `budget_tokens` with a 400, so + * sampling and thinking fields are chosen together, by model id. */ - private thinkingExtras(): Record { - const lv = this.config.reasoningLevel ?? 'auto'; - if (lv === 'auto' || lv === 'adaptive') return {}; // P54: no Anthropic equivalent of 'adaptive' - // P55: the Mythos-class models (Fable 5 / Mythos 5) run always-on adaptive thinking - // and reject an attempt to disable it — leave them on the provider default. - if (/fable|mythos/i.test(this.config.model ?? '')) return {}; - if (lv === 'off') return { thinking: { type: 'disabled' } }; - const budgets: Record = { low: 1024, medium: 4096, high: 16384 }; - const budget = Math.min(budgets[lv] ?? 4096, Math.max(1024, this.maxTokens() - 1024)); - return { thinking: { type: 'enabled', budget_tokens: budget } }; + private requestExtras(): Record { + return anthropicReasoningParams( + this.config.reasoningLevel, this.config.model, this.maxTokens(), this.config.temperature, + ); } private buildBody(messages: ChatMessage[], stream: boolean) { @@ -91,11 +87,10 @@ export class AnthropicAdapter extends BaseLLMAdapter { return { model: this.config.model, max_tokens: this.maxTokens(), - temperature: this.config.temperature ?? 0.3, system: systemPrompt, stream, messages: rest.map((m) => ({ role: m.role, content: m.content })), - ...this.thinkingExtras(), + ...this.requestExtras(), }; } @@ -120,18 +115,25 @@ export class AnthropicAdapter extends BaseLLMAdapter { * P53: POST, and if the gateway rejects the `thinking` field with a 400, retry ONCE * without it. An Anthropic-compatible proxy that predates extended thinking should * degrade to "no reasoning control", never to a dead request. + * P131: same for `output_config` (effort) and for sampling — a model id we don't + * recognise yet that has dropped `temperature` should degrade, not die. */ private async post(body: any, signal: AbortSignal): Promise { const url = `${this.getBaseURL()}/v1/messages`; const send = (b: any) => llmFetch(url, { method: 'POST', headers: this.buildHeaders(), body: JSON.stringify(b), signal, - }); + }, undefined, { connectMs: b.stream ? undefined : 0 }); const res = await send(body); - if (res.status !== 400 || !('thinking' in body)) return res; + const optional = ['thinking', 'output_config', 'temperature'].filter((k) => k in body); + if (res.status !== 400 || optional.length === 0) return res; const text = await res.clone().text(); - if (!/thinking/i.test(text)) return res; const stripped = { ...body }; - delete stripped.thinking; + if (/thinking|budget_tokens|effort|output_config/i.test(text)) { + delete stripped.thinking; + delete stripped.output_config; + } + if (/temperature|top_p|top_k|sampling/i.test(text)) delete stripped.temperature; + if (optional.every((k) => k in stripped)) return res; // the 400 is about something else return send(stripped); } @@ -162,7 +164,8 @@ export class AnthropicAdapter extends BaseLLMAdapter { .join(''); // P53: drop any inline block (OSS models behind Anthropic-compatible proxies) - return this.finalize(splitThinking(content).answer, data.usage, data.stop_reason); + const answer = withRefusal(splitThinking(content).answer, data.stop_reason, data.stop_details); + return this.finalize(answer, data.usage, data.stop_reason); } catch (err) { throw toLLMError(err, 'Anthropic 请求失败'); } finally { @@ -180,6 +183,7 @@ export class AnthropicAdapter extends BaseLLMAdapter { let acc = ''; const usage: { input_tokens?: number; output_tokens?: number } = {}; let stopReason: string | null = null; + let stopDetails: any = null; try { yield { type: 'start', requestId }; @@ -217,6 +221,7 @@ export class AnthropicAdapter extends BaseLLMAdapter { } case 'message_delta': { if (payload.delta?.stop_reason) stopReason = payload.delta.stop_reason; + if (payload.delta?.stop_details) stopDetails = payload.delta.stop_details; if (payload.usage?.output_tokens != null) usage.output_tokens = payload.usage.output_tokens; break; @@ -237,6 +242,11 @@ export class AnthropicAdapter extends BaseLLMAdapter { } } + const notice = withRefusal(acc, stopReason, stopDetails).slice(acc.length); + if (notice) { + acc += notice; + yield { type: 'delta', requestId, chunk: notice }; + } yield { type: 'done', requestId, @@ -325,15 +335,12 @@ export class AnthropicAdapter extends BaseLLMAdapter { const body: any = { model: this.config.model, max_tokens: this.maxTokens(), - temperature: this.config.temperature ?? 0.3, system: systemPrompt, stream, messages: wire, - ...this.thinkingExtras(), + ...this.requestExtras(), }; if (tools && tools.length > 0) body.tools = this.toAnthropicTools(tools); - // Extended thinking requires the default temperature - if (body.thinking?.type === 'enabled') delete body.temperature; return body; } @@ -365,7 +372,7 @@ export class AnthropicAdapter extends BaseLLMAdapter { const split = splitThinking(content); return { - content: split.answer, + content: withRefusal(split.answer, data.stop_reason, data.stop_details), reasoning: [reasoning, split.reasoning].filter(Boolean).join('\n') || undefined, toolCalls: toolCalls.length ? toolCalls : undefined, finishReason: toolCalls.length ? 'tool_use' : 'stop', @@ -401,6 +408,7 @@ export class AnthropicAdapter extends BaseLLMAdapter { let content = ''; let reasoning = ''; let stopReason: string | null = null; + let stopDetails: any = null; const usage: { input_tokens?: number; output_tokens?: number } = {}; try { @@ -473,6 +481,7 @@ export class AnthropicAdapter extends BaseLLMAdapter { case 'message_delta': if (payload.delta?.stop_reason) stopReason = payload.delta.stop_reason; + if (payload.delta?.stop_details) stopDetails = payload.delta.stop_details; if (payload.usage?.output_tokens != null) usage.output_tokens = payload.usage.output_tokens; break; @@ -485,10 +494,14 @@ export class AnthropicAdapter extends BaseLLMAdapter { } const split = splitThinking(content); + const answer = withRefusal(split.answer, stopReason, stopDetails); + if (answer.length > split.answer.length) { + yield { kind: 'text', chunk: answer.slice(split.answer.length) }; + } yield { kind: 'done', response: { - content: split.answer, + content: answer, reasoning: [reasoning, split.reasoning].filter(Boolean).join('\n') || undefined, toolCalls: toolCalls.length ? toolCalls : undefined, finishReason: toolCalls.length ? 'tool_use' @@ -573,3 +586,15 @@ export class AnthropicAdapter extends BaseLLMAdapter { }; } } + +/** + * P131: Fable 5.1 / Opus 5.5 run safety classifiers that can decline a turn — HTTP 200, + * `stop_reason: 'refusal'`, usually with no text. Without a notice the agent loop + * finished the turn with an empty bubble and no explanation. + */ +function withRefusal(text: string, stopReason: string | null, details?: { category?: string | null } | null): string { + if (stopReason !== 'refusal') return text; + const cat = details?.category ? `(${details.category})` : ''; + const notice = `⚠️ 模型的安全策略拒绝了这次请求${cat}。可以换个说法重试,或在设置里换用其他模型。`; + return text ? `${text}\n\n${notice}` : notice; +} diff --git a/src/main/llm/context-window.ts b/src/main/llm/context-window.ts index a6e7818..af8e24e 100644 --- a/src/main/llm/context-window.ts +++ b/src/main/llm/context-window.ts @@ -126,10 +126,10 @@ export function truncateMessages( ? MODEL_TOKEN_BUDGETS[budgetKey] : DEFAULT_BUDGET) - OUTPUT_RESERVE; - let availableTokens = totalBudget - estimateTokens(systemPrompt); - if (availableTokens <= 0) { - return messages.slice(-1); - } + // P131: an overhead larger than the budget used to return `messages.slice(-1)`, which can + // be a lone tool result (a 400). Fall through instead: the hard floor below keeps the last + // whole block. + let availableTokens = Math.max(0, totalBudget - estimateTokens(systemPrompt)); const blocks = toBlocks(messages); const kept: ChatMessage[][] = []; diff --git a/src/main/llm/net.ts b/src/main/llm/net.ts index 46e1824..c48c786 100644 --- a/src/main/llm/net.ts +++ b/src/main/llm/net.ts @@ -78,7 +78,10 @@ async function fetchWithConnectTimeout( connectMs = 20_000, ): Promise { const controller = new AbortController(); - const timer = setTimeout(() => controller.abort(new Error('connect timeout')), connectMs); + // P131: connectMs <= 0 → no connect-stage timer (see LlmFetchOptions.connectMs) + const timer = connectMs > 0 + ? setTimeout(() => controller.abort(new Error('connect timeout')), connectMs) + : undefined; const onAbort = () => controller.abort(signal?.reason); if (signal) { if (signal.aborted) controller.abort(signal.reason); @@ -95,6 +98,18 @@ async function fetchWithConnectTimeout( } } +export interface LlmFetchOptions { + /** + * P131: connect-stage timeout. The P109 premise — "headers arrive fast, the body is the + * outer timeout's job" — holds only for STREAMING. A non-streaming server sends headers + * when the whole answer is done, so 20 s became a total timeout for vision captions and + * non-streamed turns, and the abort counted as transient: the finished-but-slow request + * was re-sent (and billed) up to 3× per stack. Pass 0 for non-streaming requests; the + * caller's own total timeout governs them. + */ + connectMs?: number; +} + /** * Fetch that respects the OS system proxy (via Electron's Chromium stack) and * retries transient network errors. Same signature as global fetch. @@ -103,6 +118,7 @@ export async function llmFetch( url: string, init: RequestInit = {}, retries = 2, + opts: LlmFetchOptions = {}, ): Promise { let lastErr: unknown; @@ -116,7 +132,7 @@ export async function llmFetch( // 1) Undici path — no proxy, direct DNS. try { - return await fetchWithConnectTimeout(fetch, url, init, init.signal ?? undefined); + return await fetchWithConnectTimeout(fetch, url, init, init.signal ?? undefined, opts.connectMs); } catch (err) { lastErr = err; if (!isTransient(err)) { @@ -130,7 +146,7 @@ export async function llmFetch( if (typeof net !== 'undefined' && typeof net.fetch === 'function') { try { return await fetchWithConnectTimeout( - (u, i) => net.fetch(u, i as any), url, init, init.signal ?? undefined, + (u, i) => net.fetch(u, i as any), url, init, init.signal ?? undefined, opts.connectMs, ); } catch (err) { lastErr = err; diff --git a/src/main/llm/openai.ts b/src/main/llm/openai.ts index 09d7be4..4a0e43e 100644 --- a/src/main/llm/openai.ts +++ b/src/main/llm/openai.ts @@ -20,7 +20,10 @@ import { extractFirstCodeBlock } from './code-extract'; import { LLMHttpError, extractErrorMessage, toLLMError } from './errors'; import { parseSSE } from './sse'; import { buildOpenAITools } from './tools-schema'; -import { ThinkSplitter, reasoningParams, isReasoningParamError, splitThinking, dropsTemperature, detectDialect } from './thinking'; +import { + ThinkSplitter, reasoningParams, isReasoningParamError, splitThinking, dropsTemperature, detectDialect, + isOpenAIReasoningModel, isGpt6Family, +} from './thinking'; import type { ChatMessage, LLMResponse, @@ -54,12 +57,24 @@ interface PartialCall { export class OpenAIAdapter extends BaseLLMAdapter { /** P51: extra body fields for reasoning depth — empty when set to 'auto'. */ - private reasoningExtras(): Record { - return reasoningParams( + private reasoningExtras(withTools = false): Record { + const extras = reasoningParams( this.config.reasoningLevel, this.config.reasoningDialect, this.config.baseURL, ); + // P131: GPT-6 on /chat/completions 400s on reasoning_effort + tools, and on the + // 'minimal' effort we map 'off' to. Agent turns always carry tools, so they run at + // the model's default effort; plain chat gets the closest accepted level. + if ('reasoning_effort' in extras && isGpt6Family(this.config.model)) { + if (withTools) { + const rest = { ...extras }; + delete rest.reasoning_effort; + return rest; + } + if (extras.reasoning_effort === 'minimal') return { ...extras, reasoning_effort: 'low' }; + } + return extras; } // P54: 8192 truncated answers on reasoning models — the scratchpad is spent from the @@ -83,11 +98,11 @@ export class OpenAIAdapter extends BaseLLMAdapter { * 400 and require `max_completion_tokens`; every other OpenAI-compatible gateway * still expects `max_tokens`. Detect by host + model id rather than sending both, * because strict gateways reject unknown fields. + * P131: the pattern only knew gpt-5 — GPT-6 Astra got max_tokens + temperature and + * every request 400'd. */ private isOpenAIReasoner(): boolean { - const host = (this.config.baseURL ?? '').toLowerCase(); - if (!host.includes('openai.com') && !host.includes('azure.com')) return false; - return /^(o\d|gpt-5)/i.test(this.config.model ?? ''); + return isOpenAIReasoningModel(this.config.baseURL, this.config.model); } private tokenCap(): Record { @@ -235,10 +250,12 @@ export class OpenAIAdapter extends BaseLLMAdapter { const res = await llmFetch(`${this.getBaseURL()}/chat/completions`, { method: 'POST', headers: this.buildHeaders(), + // P131: reasoning models reject `max_tokens` — "Test connection" failed for + // every GPT-5.x / GPT-6 model on api.openai.com even with a valid key. body: JSON.stringify({ model: this.config.model, messages: [{ role: 'user', content: 'hi' }], - max_tokens: 1, + ...(this.isOpenAIReasoner() ? { max_completion_tokens: 32 } : { max_tokens: 1 }), }), signal: s, }); @@ -267,7 +284,7 @@ export class OpenAIAdapter extends BaseLLMAdapter { tool_choice: 'auto', stream, ...(stream ? { stream_options: { include_usage: true } } : {}), - ...this.reasoningExtras(), + ...this.reasoningExtras(true), }; } @@ -358,7 +375,7 @@ export class OpenAIAdapter extends BaseLLMAdapter { const url = `${this.getBaseURL()}/chat/completions`; const send = (b: any) => llmFetch(url, { method: 'POST', headers: this.buildHeaders(), body: JSON.stringify(b), signal, - }); + }, undefined, { connectMs: b.stream ? undefined : 0 }); const extras = this.reasoningExtras(); const res = await send(body); diff --git a/src/main/llm/thinking.ts b/src/main/llm/thinking.ts index e964213..9369950 100644 --- a/src/main/llm/thinking.ts +++ b/src/main/llm/thinking.ts @@ -183,6 +183,132 @@ export function reasoningParams( } } +// ===== P131: model capability detection ===== + +/** + * OpenAI's own reasoning models (o-series, GPT-5.x, GPT-6.x) reject `max_tokens` + * (they need `max_completion_tokens`) and reject `temperature` / `top_p`. Only on + * OpenAI's / Azure's own hosts — every other OpenAI-compatible gateway still expects + * `max_tokens`, and strict gateways reject unknown fields. + */ +export function isOpenAIReasoningModel(baseURL?: string, model?: string): boolean { + const host = (baseURL ?? '').toLowerCase(); + if (!host.includes('openai.com') && !host.includes('azure.com')) return false; + return /^(o\d|gpt-([5-9]|\d{2,}))/i.test(model ?? ''); +} + +/** + * P131: GPT-6 (Astra and later) on /chat/completions rejects `reasoning_effort` whenever + * `tools` are present ("Function tools with reasoning_effort are not supported … + * use /v1/responses"), and rejects the 'none' / 'minimal' efforts outright. + */ +export function isGpt6Family(model?: string): boolean { + return /^gpt-([6-9]|\d{2,})/i.test(model ?? ''); +} + +/** What a Claude model's request surface accepts. */ +export interface ClaudeCaps { + /** temperature / top_p / top_k are rejected with a 400 — never send them. */ + noSampling: boolean; + /** Reasoning depth is `thinking: {type:'adaptive'}` + `output_config.effort`; + * `{type:'enabled', budget_tokens}` is removed or deprecated. */ + adaptive: boolean; + /** Thinking runs even when `thinking` is omitted. */ + thinksByDefault: boolean; + /** Thinking cannot be switched off — `{type:'disabled'}` is a 400. */ + alwaysThinks: boolean; +} + +/** + * P131: request-surface capabilities by Claude model id. Matches first-party ids + * (`claude-opus-5-5`), dotted proxy ids (`claude-opus-4.8`) and prefixed ids + * (`anthropic.claude-fable-5-1`). Unrecognised ids — old dated models, OSS models + * behind Anthropic-compatible proxies — get the legacy surface. + * + * Fable 5 / 5.1, Mythos always thinks; sampling + budget_tokens → 400 + * Opus 5.5 always thinks; sampling + budget_tokens → 400 + * Sonnet 5.5 disabled → 400; non-default sampling → 400 + * Opus 5, Sonnet 5 thinks by default; sampling + budget_tokens → 400 + * Opus 4.7 / 4.8 adaptive only on-mode (off by default); sampling → 400 + * Opus 4.6 / Sonnet 4.6 adaptive recommended; sampling allowed + * older (Haiku 4.5, …) budget_tokens thinking; sampling allowed + */ +export function claudeCaps(model?: string): ClaudeCaps { + const legacy: ClaudeCaps = { noSampling: false, adaptive: false, thinksByDefault: false, alwaysThinks: false }; + const id = (model ?? '').toLowerCase(); + if (/fable|mythos/.test(id)) { + return { noSampling: true, adaptive: true, thinksByDefault: true, alwaysThinks: true }; + } + const m = /(opus|sonnet|haiku)[-_ ]?(\d{1,2})(?!\d)(?:[-.](\d)(?!\d))?/.exec(id); + if (!m) return legacy; + const family = m[1]; + const v = Number(m[2]) + Number(m[3] ?? 0) / 10; + if (family === 'opus') { + if (v >= 5.5) return { noSampling: true, adaptive: true, thinksByDefault: true, alwaysThinks: true }; + if (v >= 5) return { noSampling: true, adaptive: true, thinksByDefault: true, alwaysThinks: false }; + if (v >= 4.7) return { noSampling: true, adaptive: true, thinksByDefault: false, alwaysThinks: false }; + if (v >= 4.6) return { noSampling: false, adaptive: true, thinksByDefault: false, alwaysThinks: false }; + return legacy; + } + if (family === 'sonnet') { + if (v >= 5.5) return { noSampling: true, adaptive: true, thinksByDefault: true, alwaysThinks: true }; + if (v >= 5) return { noSampling: true, adaptive: true, thinksByDefault: true, alwaysThinks: false }; + if (v >= 4.6) return { noSampling: false, adaptive: true, thinksByDefault: false, alwaysThinks: false }; + return legacy; + } + return legacy; +} + +/** + * P131: Anthropic request fields for reasoning depth + sampling, by model. + * + * Before P131 every request carried `temperature: 0.3` — a hard 400 on Fable 5 / 5.1, + * Opus 4.7+ and Sonnet 5+ (every Claude preset except Sonnet 4.6) — and reasoning + * levels were sent as `budget_tokens`, which those models also reject. + * + * Adaptive-surface models: level → `output_config.effort`; thinking summaries are + * requested (`display: 'summarized'`) whenever thinking runs, because the default + * ('omitted') streams empty thinking blocks and the UI would show nothing while the + * model works. 'off' on a model that cannot stop thinking becomes the lowest effort. + */ +export function anthropicReasoningParams( + level: ReasoningLevel | undefined, + model: string | undefined, + maxTokens: number, + temperature: number | undefined, +): Record { + const lv = level ?? 'auto'; + const caps = claudeCaps(model); + const adaptiveOn = { thinking: { type: 'adaptive', display: 'summarized' } }; + + if (caps.adaptive) { + let out: Record; + if (lv === 'off') { + out = caps.alwaysThinks ? { ...adaptiveOn, output_config: { effort: 'low' } } : { thinking: { type: 'disabled' } }; + } else if (lv === 'low' || lv === 'medium' || lv === 'high') { + out = { ...adaptiveOn, output_config: { effort: lv } }; + } else if (lv === 'adaptive' || caps.thinksByDefault) { + // 'auto' on a model that thinks anyway: same behaviour, readable summaries. + out = { ...adaptiveOn }; + } else { + out = {}; // 'auto' on Opus 4.6–4.8 / Sonnet 4.6: provider default (no thinking) + } + // Sampling is rejected outright on noSampling models, and incompatible with thinking on. + if (!caps.noSampling && (!out.thinking || out.thinking.type === 'disabled')) { + out.temperature = temperature ?? 0.3; + } + return out; + } + + // Legacy surface: budget_tokens thinking; temperature only while thinking is off. + // P55: an unrecognised Mythos-class id is handled above (always thinks). + if (lv === 'auto' || lv === 'adaptive') return { temperature: temperature ?? 0.3 }; + if (lv === 'off') return { temperature: temperature ?? 0.3, thinking: { type: 'disabled' } }; + const budgets: Record = { low: 1024, medium: 4096, high: 16384 }; + const budget = Math.min(budgets[lv] ?? 4096, Math.max(1024, maxTokens - 1024)); + return { thinking: { type: 'enabled', budget_tokens: budget } }; +} + /** P54: providers that ignore sampling params while thinking — we drop them to avoid noise. */ export function dropsTemperature( level: ReasoningLevel | undefined, @@ -205,7 +331,10 @@ export const REASONING_FIELDS = [ export function isReasoningParamError(body: string): boolean { const b = (body || '').toLowerCase(); if (!b) return false; + // P131: 'not supported' — GPT-6 Astra phrases its tools+reasoning_effort 400 as + // "Function tools with reasoning_effort are not supported …", which fell through. const complains = b.includes('unknown') || b.includes('unsupported') || b.includes('invalid') - || b.includes('not allowed') || b.includes('unrecognized') || b.includes('extra'); + || b.includes('not allowed') || b.includes('unrecognized') || b.includes('extra') + || b.includes('not supported'); return complains && REASONING_FIELDS.some((f) => b.includes(f)); } diff --git a/src/main/llm/vision.ts b/src/main/llm/vision.ts index da8046c..a40d38d 100644 --- a/src/main/llm/vision.ts +++ b/src/main/llm/vision.ts @@ -4,6 +4,7 @@ import type { VisionConfig, LocaleName } from '../../shared/types'; import { toLLMError } from './errors'; import { llmFetch } from './net'; +import { isOpenAIReasoningModel } from './thinking'; export interface AnalyzeImageInput { question: string; // Question drafted by the main model = image-to-text prompt @@ -41,11 +42,24 @@ export async function analyzeImage(input: AnalyzeImageInput): Promise { ], }, ], - temperature: 0.2, - max_tokens: 1024, + // P131: OpenAI reasoning models (GPT-5.x / GPT-6) as the vision model 400'd on + // max_tokens + temperature. Their cap also covers reasoning tokens, so 1024 would + // leave an empty caption — give them room. + ...(isOpenAIReasoningModel(config.baseURL, config.model) + ? { max_completion_tokens: 8192 } + : { temperature: 0.2, max_tokens: 1024 }), stream: false, }; + // P131: a caption is one non-streamed answer — its headers arrive only when it is done, so + // the connect-stage timer must not apply (it killed and re-sent slow captions at 20 s). + // VisionConfig.timeoutMs (default 120 s) is the total budget instead. + const controller = new AbortController(); + const onAbort = () => controller.abort(signal?.reason); + if (signal?.aborted) controller.abort(signal.reason); + else signal?.addEventListener('abort', onAbort, { once: true }); + const timer = setTimeout(() => controller.abort(new Error('视觉模型响应超时')), config.timeoutMs ?? 120_000); let res: Response; + let text: string; try { res = await llmFetch(`${base}/chat/completions`, { method: 'POST', @@ -54,12 +68,15 @@ export async function analyzeImage(input: AnalyzeImageInput): Promise { Authorization: `Bearer ${config.apiKey}`, }, body: JSON.stringify(body), - signal, - }); + signal: controller.signal, + }, undefined, { connectMs: 0 }); + text = await res.text(); } catch (err) { throw toLLMError(err, '视觉模型网络请求失败'); + } finally { + clearTimeout(timer); + signal?.removeEventListener('abort', onAbort); } - const text = await res.text(); if (!res.ok) { throw toLLMError(new Error(text), `视觉模型请求失败 (HTTP ${res.status})`); } diff --git a/src/main/store/config.ts b/src/main/store/config.ts index 11ba15e..a795c9d 100644 --- a/src/main/store/config.ts +++ b/src/main/store/config.ts @@ -25,6 +25,10 @@ export interface StoredConfig { const SCHEMA_VERSION = 1; +// P131: the API key currently supplied by the environment (.env / process.env). It is used, +// never persisted — saveConfig drops it so a later save cannot copy it into the store. +let envSourcedKey = ''; + // P107: live approval-mode cache — updated on every saveConfig so an in-flight agent // session can re-read it per tool call (issue #1: settings changes should apply to // the running conversation immediately). @@ -99,7 +103,18 @@ export async function loadConfig(): Promise { console.info( `[Millwright] 使用 .env fallback 配置: protocol=${envFallback.protocol}, model=${envFallback.model}`, ); - return envFallback; + // P131: take only the CONNECTION from the environment. Returning the env config whole + // threw away every saved preference (approval mode, disabled tools, context window…), + // and the next save then encrypted the env key into the store. + envSourcedKey = envFallback.apiKey; + return { + ...DEFAULT_CONFIG, + ...llm, + protocol: envFallback.protocol, + baseURL: envFallback.baseURL, + model: envFallback.model, + apiKey: envFallback.apiKey, + }; } } @@ -119,7 +134,7 @@ export async function saveConfig(config: LLMConfig): Promise { const { apiKey, ...rest } = config; let encryptedApiKey = ''; - if (apiKey) { + if (apiKey && apiKey !== envSourcedKey) { if (safeStorage.isEncryptionAvailable()) { encryptedApiKey = safeStorage.encryptString(apiKey).toString('base64'); } else { diff --git a/src/preload/index.ts b/src/preload/index.ts index 96eade7..324a01a 100644 --- a/src/preload/index.ts +++ b/src/preload/index.ts @@ -28,12 +28,12 @@ const api = { save: (config: LLMConfig) => ipcRenderer.invoke(IpcChannels.CONFIG_SAVE, config), }, theme: { - load: (): Promise => ipcRenderer.invoke('theme:load'), - save: (theme: ThemeName) => ipcRenderer.invoke('theme:save', theme), + load: (): Promise => ipcRenderer.invoke(IpcChannels.THEME_LOAD), + save: (theme: ThemeName) => ipcRenderer.invoke(IpcChannels.THEME_SAVE, theme), }, locale: { - load: (): Promise => ipcRenderer.invoke('locale:load'), - save: (locale: LocaleName) => ipcRenderer.invoke('locale:save', locale), + load: (): Promise => ipcRenderer.invoke(IpcChannels.LOCALE_LOAD), + save: (locale: LocaleName) => ipcRenderer.invoke(IpcChannels.LOCALE_SAVE, locale), }, sw: { connect: (): Promise<{ ok: boolean; status: SWStatus }> => diff --git a/src/renderer/App.tsx b/src/renderer/App.tsx index b2bf66d..931f453 100644 --- a/src/renderer/App.tsx +++ b/src/renderer/App.tsx @@ -35,8 +35,17 @@ export default function App() { // —— Config —— const [config, setConfig] = useState(DEFAULT_CONFIG); + const configLoaded = useRef(false); useEffect(() => { - window.api.config.load().then(setConfig); + window.api.config.load() + .then((c) => { configLoaded.current = true; setConfig(c); }) + .catch((err) => console.error('[config] load failed', err)); + }, []); + // P131: never persist before the stored config has loaded — an approval or tools toggle + // in that window saved DEFAULT_CONFIG over the real one and wiped the API key. + const persistConfig = useCallback((next: LLMConfig) => { + setConfig(next); + if (configLoaded.current) void window.api.config.save(next); }, []); // —— SolidWorks status —— @@ -322,11 +331,7 @@ export default function App() { onCancel={cancel} isGenerating={isGenerating} approvalMode={config.approvalMode ?? 'normal'} - onApprovalChange={(m) => { - const next = { ...config, approvalMode: m }; - setConfig(next); - void window.api.config.save(next); - }} + onApprovalChange={(m) => persistConfig({ ...config, approvalMode: m })} placeholder={ !config.apiKey ? tr('input.placeholderNoKey') @@ -343,11 +348,7 @@ export default function App() { { - const cfg = { ...config, disabledTools: next }; - setConfig(cfg); - window.api.config.save(cfg); - }} + onChange={(next) => persistConfig({ ...config, disabledTools: next })} /> )} diff --git a/src/renderer/components/SettingsModal.tsx b/src/renderer/components/SettingsModal.tsx index 159b6be..21f24a0 100644 --- a/src/renderer/components/SettingsModal.tsx +++ b/src/renderer/components/SettingsModal.tsx @@ -75,11 +75,27 @@ export function SettingsModal({ }; const handleProtocol = (p: 'anthropic' | 'openai') => { + // P131: pair the default URL with a model that endpoint actually serves. The first + // OpenAI-protocol preset is DeepSeek, so switching gave api.openai.com + deepseek-v4-pro. + const official = OPENAI_COMPATIBLE_PROVIDERS.find((x) => x.url === DEFAULT_URLS[p]); setDraft((d) => ({ ...d, protocol: p, baseURL: DEFAULT_URLS[p], - model: MODEL_PRESETS[p][0].value, + model: official?.suggestedModel ?? MODEL_PRESETS[p][0].value, + })); + setTestStatus({ kind: 'idle' }); + }; + + // P131: a quick-fill button sets the whole provider, not just its URL — the suggested + // model and the per-provider context / output defaults were declared but never applied. + const applyProvider = (p: (typeof OPENAI_COMPATIBLE_PROVIDERS)[number]) => { + setDraft((d) => ({ + ...d, + baseURL: p.url, + ...(p.suggestedModel ? { model: p.suggestedModel } : {}), + ...(p.contextWindow ? { contextWindow: p.contextWindow } : {}), + ...(p.maxTokens ? { maxTokens: p.maxTokens } : {}), })); setTestStatus({ kind: 'idle' }); }; @@ -243,7 +259,7 @@ export function SettingsModal({ > {tr('settings.swConnection')}
- + (
); } + +/** + * P131: a number field that clamps when editing ends (blur / Enter), not on every keystroke. + * Clamping per keystroke snapped a half-typed "2" to the minimum, so a value such as 200000 + * could not be typed at all. + */ +function ClampedNumber({ value, min, max, step, fallback, onCommit, style }: { + value: number; + min: number; + max: number; + step?: number; + fallback: number; + onCommit: (v: number) => void; + style?: React.CSSProperties; +}) { + const [text, setText] = useState(String(value)); + useEffect(() => { setText(String(value)); }, [value]); + const commit = () => { + const n = Number(text); + const v = Math.max(min, Math.min(max, text.trim() !== '' && Number.isFinite(n) ? Math.round(n) : fallback)); + setText(String(v)); + if (v !== value) onCommit(v); + }; + return ( + setText(e.target.value)} + onBlur={commit} + onKeyDown={(e) => { if (e.key === 'Enter') commit(); }} + style={style} + /> + ); +} diff --git a/src/renderer/components/Sidebar.tsx b/src/renderer/components/Sidebar.tsx index e51d56d..8cee572 100644 --- a/src/renderer/components/Sidebar.tsx +++ b/src/renderer/components/Sidebar.tsx @@ -98,7 +98,7 @@ export function Sidebar({ }} >
- + {target}} - {name} + + {name}{typeof s.durationMs === 'number' ? ` · ${(s.durationMs / 1000).toFixed(1)}s` : ''} + {hasDetail && {rowOpen ? '▲' : '▼'}} {rowOpen && hasDetail && ( diff --git a/src/renderer/hooks/useLLM.ts b/src/renderer/hooks/useLLM.ts index 60f81c6..f2b6a20 100644 --- a/src/renderer/hooks/useLLM.ts +++ b/src/renderer/hooks/useLLM.ts @@ -135,11 +135,11 @@ export function useLLM({ config, initial }: UseLLMOptions) { for (let i = steps.length - 1; i >= 0; i--) { const s = steps[i]; if (s.kind === 'tool' && (s.id === key || s.name === tc?.name) && s.status === 'running') { - steps[i] = { ...s, status, result: tc?.result, params: s.params ?? tc?.parameters }; + steps[i] = { ...s, status, result: tc?.result, params: s.params ?? tc?.parameters, durationMs: tc?.durationMs }; return { ...m, steps }; } } - steps.push({ kind: 'tool', id: key, name: tc?.name, params: tc?.parameters, status, result: tc?.result }); + steps.push({ kind: 'tool', id: key, name: tc?.name, params: tc?.parameters, status, result: tc?.result, durationMs: tc?.durationMs }); return { ...m, steps }; }); }, [updateAssistant]); diff --git a/src/renderer/session-export.ts b/src/renderer/session-export.ts index 7d2d8a8..f91c439 100644 --- a/src/renderer/session-export.ts +++ b/src/renderer/session-export.ts @@ -50,7 +50,7 @@ function statusMark(status?: string): string { function toolToMarkdown(s: AgentStep): string { // P130: surface wall-clock duration next to the tool name so a long gear/batch is // visible at a glance ("- ✓ `extrude` · 1.2s"). - const dur = (s as any).durationMs; + const dur = s.durationMs; const durStr = typeof dur === 'number' ? ` · ${(dur / 1000).toFixed(1)}s` : ''; const lines = [`- ${statusMark(s.status)} \`${s.name ?? 'tool'}\`${durStr}`]; if (s.params && Object.keys(s.params).length) { @@ -135,7 +135,7 @@ export function sessionToJSON(messages: ChatMessage[]): string { ...(s.status ? { status: s.status } : {}), ...(s.result ? { result: s.result } : {}), // P130: persist the wall-clock duration alongside the tool step. - ...((s as any).durationMs !== undefined ? { durationMs: (s as any).durationMs } : {}), + ...(s.durationMs !== undefined ? { durationMs: s.durationMs } : {}), })), } : {}), diff --git a/src/shared/ipc-channels.ts b/src/shared/ipc-channels.ts index cc3114e..baf0853 100644 --- a/src/shared/ipc-channels.ts +++ b/src/shared/ipc-channels.ts @@ -50,6 +50,12 @@ export const IpcChannels = { UPDATE_SKIP: 'update:skip', UPDATE_OPEN_RELEASE: 'update:open-release', UPDATE_EVENT: 'update:event', + + // Theme / UI language (kept independent of the LLM config) + THEME_LOAD: 'theme:load', + THEME_SAVE: 'theme:save', + LOCALE_LOAD: 'locale:load', + LOCALE_SAVE: 'locale:save', } as const; export type IpcChannel = (typeof IpcChannels)[keyof typeof IpcChannels]; diff --git a/src/shared/presets.ts b/src/shared/presets.ts index 6f260ad..ad0c2a7 100644 --- a/src/shared/presets.ts +++ b/src/shared/presets.ts @@ -2,6 +2,9 @@ // Model presets / default URLs / default parameters. // P54/P55: preset IDs and contextWindow/maxTokens aligned with current provider docs (2026-07). // DeepSeek's old `deepseek-chat` / `deepseek-reasoner` aliases were retired 2026-07-24. +// P131: GPT-6 Astra (2026-09-03), Claude Opus 5.5 and Claude Fable 5.1 become the +// recommended OpenAI / Anthropic models. Saved ids that left the list still load — the +// Settings dropdown shows any non-preset id as "Custom model". import type { LLMProtocol, ModelPreset, LLMConfig } from './types'; @@ -12,8 +15,8 @@ export const DEFAULT_URLS: Record = { export const MODEL_PRESETS: Record = { anthropic: [ - { label: 'Claude Fable 5 (1M ctx, always-on thinking)', value: 'claude-fable-5' }, - { label: 'Claude Opus 4.8', value: 'claude-opus-4-8' }, + { label: 'Claude Opus 5.5 (1M ctx, recommended)', value: 'claude-opus-5-5' }, + { label: 'Claude Fable 5.1 (1M ctx, most capable)', value: 'claude-fable-5-1' }, { label: 'Claude Sonnet 4.6', value: 'claude-sonnet-4-6' }, { label: 'Custom model', value: 'custom' }, ], @@ -26,6 +29,7 @@ export const MODEL_PRESETS: Record = { { label: 'Qwen 3.7 Max (阿里百炼)', value: 'qwen3.7-max' }, { label: 'GLM-4.6 (智谱)', value: 'glm-4.6' }, // —— OpenAI official —— + { label: 'GPT-6 Astra (1M ctx, recommended)', value: 'gpt-6-astra' }, { label: 'GPT-5.6 Sol', value: 'gpt-5.6-sol' }, { label: 'GPT-4.1', value: 'gpt-4.1' }, { label: 'GPT-4o Mini', value: 'gpt-4o-mini' }, @@ -50,8 +54,8 @@ export const OPENAI_COMPATIBLE_PROVIDERS: Array<{ /** P54: per-provider recommended max output (tokens). Used as the default in Settings. */ maxTokens?: number; }> = [ - // OpenAI official — GPT-5.x / o series REQUIRE max_completion_tokens (handled in adapter) - { name: 'OpenAI', url: 'https://api.openai.com/v1', supportsTools: true, suggestedModel: 'gpt-5.6-sol', contextWindow: 1_000_000, maxTokens: 32_768 }, + // OpenAI official — GPT-6 / GPT-5.x / o series REQUIRE max_completion_tokens (handled in adapter) + { name: 'OpenAI', url: 'https://api.openai.com/v1', supportsTools: true, suggestedModel: 'gpt-6-astra', contextWindow: 1_050_000, maxTokens: 32_768 }, { name: 'DeepSeek', url: 'https://api.deepseek.com', supportsTools: true, suggestedModel: 'deepseek-v4-pro', contextWindow: 1_048_576, maxTokens: 32_768 }, { name: 'Kimi / Moonshot', url: 'https://api.moonshot.cn/v1', supportsTools: true, suggestedModel: 'kimi-k2.5', contextWindow: 262_144, maxTokens: 32_768 }, { name: 'MiniMax', url: 'https://api.minimax.io/v1', supportsTools: true, suggestedModel: 'minimax-m3', contextWindow: 512_000, maxTokens: 32_768 }, diff --git a/src/shared/types.ts b/src/shared/types.ts index a6738a7..3d2188b 100644 --- a/src/shared/types.ts +++ b/src/shared/types.ts @@ -107,6 +107,9 @@ export interface AgentStep { status?: 'running' | 'ok' | 'error' | 'rejected'; /** tool step: formatted result / error message */ result?: string; + /** P131: tool step: wall-clock duration of the sidecar call (P130 computed it, but it + * was dropped before reaching the step, so the UI and exports never showed it) */ + durationMs?: number; /** P28: confirm step's parent agent requestId (for click → IPC 回执) */ requestId?: string; /** P51: 该推理块是否还在流式接收中 */ diff --git a/tests/agent-loop.test.mjs b/tests/agent-loop.test.mjs index 275e486..01c7389 100644 --- a/tests/agent-loop.test.mjs +++ b/tests/agent-loop.test.mjs @@ -40,3 +40,28 @@ test('looksLikeQuestion: asking for parameters ends the turn', async () => { assert.equal(looksLikeQuestion('What module and tooth count do you want?'), true); assert.equal(looksLikeQuestion('Plan: top plane, 80×50 plate, extrude 10 mm, then four M6 holes. Starting now.'), false); }); + +// P131: P130's pattern matched generic words anywhere (which / 多少 / 几个 / 哪个 / 你想), so +// ordinary plans read as questions (nudge skipped → user must type "继续" again), and +// common parameter requests without a question mark were missed. +test('looksLikeQuestion: plans that mention which/几个/哪个 still proceed', async () => { + const { looksLikeQuestion } = await import('../dist/main/main/agent/agent-loop-sidecar.js'); + const plans = [ + 'Plan: I will sketch an 80×50 rectangle on the Top Plane, which is the XZ plane in SolidWorks, extrude it 10 mm, then cut four Ø6.6 through holes at (±30, ±15). Starting now.', + '方案:先确定几个关键尺寸——板 80×50×10,四角 M6 通孔距边 10 mm。步骤:1) 上视基准面画矩形并拉伸 10;2) 在顶面画 4 个圆并切除。现在开始执行。', + '整体方案如下:齿轮模数 2、齿数 20、齿宽 15、内孔 Ø10。先说明一下在哪个基准面建模:用前视基准面,这样齿轮轴线沿 Z。接下来调用 create_spur_gear 生成齿轮,再切键槽。', + '根据你想要的规格(法兰外径 120、厚 12、6 个 M8 孔均布在 PCD 90 上),方案:1) 前视基准面画 Ø120 圆拉伸 12;2) 顶面画 Ø9 孔,圆周阵列 6 个;3) 中心孔 Ø40。开始建模。', + '方案:壁厚统一 2 mm,不管外形多少个圆角,最后统一 fillet R1。先画 100×60 轮廓拉伸 30,再 shell 2 mm 开口朝上,最后倒角。现在开始。', + ]; + for (const p of plans) assert.equal(looksLikeQuestion(p), false, p); +}); + +test('looksLikeQuestion: parameter requests without a question mark end the turn', async () => { + const { looksLikeQuestion } = await import('../dist/main/main/agent/agent-loop-sidecar.js'); + const asks = [ + '要生成这个齿轮,我还缺少以下参数:\n1. 模数 m\n2. 齿数 z\n3. 齿宽\n4. 内孔直径\n收到后我会立即建模。', + '为了准确建模,请补充下列尺寸:法兰外径、厚度、螺栓孔数量与孔径、分布圆直径。拿到这些数值后我会一次性完成建模,不会使用示例值代替。', + 'Before I build the gear I need a few values from you: the module, the number of teeth, the face width and the bore diameter. Once I have them I will create it in one go.', + ]; + for (const a of asks) assert.equal(looksLikeQuestion(a), true, a); +}); diff --git a/tests/llm-request.test.mjs b/tests/llm-request.test.mjs new file mode 100644 index 0000000..7de541e --- /dev/null +++ b/tests/llm-request.test.mjs @@ -0,0 +1,220 @@ +// tests/llm-request.test.mjs +// +// P131: the request bodies the adapters actually put on the wire, per model. +// Current Claude models (Fable 5 / 5.1, Opus 4.7+, Sonnet 5+) reject `temperature` +// and `budget_tokens` with a 400; GPT-6 Astra rejects `max_tokens`, `temperature` +// and `reasoning_effort` alongside tools. A stubbed global fetch records each body. + +import { test, afterEach } from 'node:test'; +import assert from 'node:assert/strict'; +import { AnthropicAdapter } from '../dist/main/main/llm/anthropic.js'; +import { OpenAIAdapter } from '../dist/main/main/llm/openai.js'; +import { + claudeCaps, + isOpenAIReasoningModel, + isReasoningParamError, +} from '../dist/main/main/llm/thinking.js'; + +const realFetch = globalThis.fetch; +afterEach(() => { globalThis.fetch = realFetch; }); + +/** Stub fetch: answers each request with the next queued [status, json] pair. */ +function stubFetch(...replies) { + const bodies = []; + globalThis.fetch = async (_url, init) => { + bodies.push(JSON.parse(init.body)); + const [status, json] = replies.length > 1 ? replies.shift() : replies[0]; + return new Response(JSON.stringify(json), { + status, headers: { 'content-type': 'application/json' }, + }); + }; + return bodies; +} + +const CLAUDE_OK = { content: [{ type: 'text', text: 'ok' }], stop_reason: 'end_turn' }; +const OPENAI_OK = { choices: [{ message: { content: 'ok' }, finish_reason: 'stop' }] }; +const MSGS = [{ role: 'user', content: 'hi' }]; +const TOOLS = [{ type: 'function', function: { name: 'sw_status', description: 'x', parameters: { type: 'object', properties: {} } } }]; + +const claude = (model, extra = {}) => new AnthropicAdapter({ + protocol: 'anthropic', baseURL: 'https://api.anthropic.com', apiKey: 'k', model, temperature: 0.3, ...extra, +}); +const openai = (model, extra = {}) => new OpenAIAdapter({ + protocol: 'openai', baseURL: 'https://api.openai.com/v1', apiKey: 'k', model, temperature: 0.3, ...extra, +}); + +test('claudeCaps: model families', () => { + for (const id of ['claude-fable-5-1', 'claude-fable-5', 'claude-mythos-5-1', 'claude-opus-5-5', 'claude-sonnet-5-5']) { + assert.equal(claudeCaps(id).alwaysThinks, true, id); + assert.equal(claudeCaps(id).noSampling, true, id); + } + for (const id of ['claude-opus-5', 'claude-opus-4-8', 'claude-opus-4.7', 'anthropic.claude-sonnet-5']) { + assert.equal(claudeCaps(id).noSampling, true, id); + assert.equal(claudeCaps(id).alwaysThinks, false, id); + } + assert.equal(claudeCaps('claude-opus-4-8').thinksByDefault, false); + assert.equal(claudeCaps('claude-opus-5').thinksByDefault, true); + assert.deepEqual(claudeCaps('claude-sonnet-4-6'), { noSampling: false, adaptive: true, thinksByDefault: false, alwaysThinks: false }); + for (const id of ['claude-haiku-4-5-20251001', 'claude-3-7-sonnet-20250219', 'claude-sonnet-4-20250514', 'MiniMax-M3', '']) { + assert.equal(claudeCaps(id).adaptive, false, id); + assert.equal(claudeCaps(id).noSampling, false, id); + } +}); + +test('anthropic: Fable 5.1 / Opus 5.5 never get temperature or budget thinking', async () => { + for (const model of ['claude-fable-5-1', 'claude-opus-5-5']) { + for (const reasoningLevel of ['auto', 'off', 'adaptive', 'low', 'medium', 'high']) { + const bodies = stubFetch([200, CLAUDE_OK]); + await claude(model, { reasoningLevel }).chatWithTools(MSGS, undefined, TOOLS); + const b = bodies[0]; + assert.equal('temperature' in b, false, `${model}/${reasoningLevel} sent temperature`); + assert.notEqual(b.thinking?.type, 'disabled', `${model}/${reasoningLevel} disabled thinking`); + assert.equal(b.thinking?.budget_tokens, undefined, `${model}/${reasoningLevel} sent budget_tokens`); + assert.equal('tool_choice' in b, false, 'forced tool_choice is a 400 on these models'); + } + } +}); + +test('anthropic: reasoning level maps to adaptive thinking + effort', async () => { + let bodies = stubFetch([200, CLAUDE_OK]); + await claude('claude-opus-5-5', { reasoningLevel: 'high' }).chat(MSGS); + assert.deepEqual(bodies[0].thinking, { type: 'adaptive', display: 'summarized' }); + assert.deepEqual(bodies[0].output_config, { effort: 'high' }); + + bodies = stubFetch([200, CLAUDE_OK]); + await claude('claude-opus-5-5', { reasoningLevel: 'off' }).chat(MSGS); + assert.deepEqual(bodies[0].output_config, { effort: 'low' }, 'off → lowest effort on an always-thinking model'); + + bodies = stubFetch([200, CLAUDE_OK]); + await claude('claude-opus-4-8', { reasoningLevel: 'off' }).chat(MSGS); + assert.deepEqual(bodies[0].thinking, { type: 'disabled' }); + assert.equal('temperature' in bodies[0], false); + + bodies = stubFetch([200, CLAUDE_OK]); + await claude('claude-opus-4-8').chat(MSGS); + assert.equal('thinking' in bodies[0], false, 'auto on Opus 4.8 = provider default'); + assert.equal('temperature' in bodies[0], false); +}); + +test('anthropic: older / proxied models keep temperature and budget thinking', async () => { + let bodies = stubFetch([200, CLAUDE_OK]); + await claude('claude-sonnet-4-6').chat(MSGS); + assert.equal(bodies[0].temperature, 0.3); + + bodies = stubFetch([200, CLAUDE_OK]); + await claude('claude-haiku-4-5', { reasoningLevel: 'medium' }).chatWithTools(MSGS, undefined, TOOLS); + assert.deepEqual(bodies[0].thinking, { type: 'enabled', budget_tokens: 4096 }); + assert.equal('temperature' in bodies[0], false, 'budget thinking requires the default temperature'); + + bodies = stubFetch([200, CLAUDE_OK]); + await claude('MiniMax-M3', { baseURL: 'https://api.minimax.io/anthropic' }).chat(MSGS); + assert.equal(bodies[0].temperature, 0.3); +}); + +test('anthropic: a 400 naming temperature is retried once without it', async () => { + const bodies = stubFetch( + [400, { type: 'error', error: { type: 'invalid_request_error', message: 'temperature: is not supported for this model' } }], + [200, CLAUDE_OK], + ); + const r = await claude('claude-unknown-9').chat(MSGS); + assert.equal(bodies.length, 2); + assert.equal(bodies[0].temperature, 0.3); + assert.equal('temperature' in bodies[1], false); + assert.equal(r.content, 'ok'); +}); + +test('anthropic: an unrelated 400 is not retried', async () => { + const bodies = stubFetch([400, { type: 'error', error: { message: 'messages: roles must alternate' } }]); + await assert.rejects(claude('claude-sonnet-4-6').chat(MSGS)); + assert.equal(bodies.length, 1); +}); + +test('anthropic: refusal stop_reason surfaces a notice instead of an empty reply', async () => { + stubFetch([200, { content: [], stop_reason: 'refusal', stop_details: { type: 'refusal', category: 'cyber' } }]); + const r = await claude('claude-fable-5-1').chatWithTools(MSGS, undefined, TOOLS); + assert.match(r.content, /拒绝/); + assert.match(r.content, /cyber/); +}); + +test('openai: GPT-6 Astra is a reasoning model on OpenAI hosts only', () => { + assert.equal(isOpenAIReasoningModel('https://api.openai.com/v1', 'gpt-6-astra'), true); + assert.equal(isOpenAIReasoningModel('https://api.openai.com/v1', 'gpt-5.6-sol'), true); + assert.equal(isOpenAIReasoningModel('https://api.openai.com/v1', 'o4-mini'), true); + assert.equal(isOpenAIReasoningModel('https://api.openai.com/v1', 'gpt-4.1'), false); + assert.equal(isOpenAIReasoningModel('https://api.siliconflow.cn/v1', 'gpt-6-astra'), false); +}); + +test('openai: GPT-6 Astra gets max_completion_tokens and no temperature', async () => { + const bodies = stubFetch([200, OPENAI_OK]); + await openai('gpt-6-astra').chatWithTools(MSGS, undefined, TOOLS); + const b = bodies[0]; + assert.ok(b.max_completion_tokens > 0); + assert.equal('max_tokens' in b, false); + assert.equal('temperature' in b, false); +}); + +test('openai: GPT-6 Astra drops reasoning_effort when tools are present', async () => { + let bodies = stubFetch([200, OPENAI_OK]); + await openai('gpt-6-astra', { reasoningLevel: 'high' }).chatWithTools(MSGS, undefined, TOOLS); + assert.equal('reasoning_effort' in bodies[0], false); + + bodies = stubFetch([200, OPENAI_OK]); + await openai('gpt-6-astra', { reasoningLevel: 'off' }).chat(MSGS); + assert.equal(bodies[0].reasoning_effort, 'low', "'minimal' is rejected by GPT-6"); + + bodies = stubFetch([200, OPENAI_OK]); + await openai('gpt-5.6-sol', { reasoningLevel: 'high' }).chatWithTools(MSGS, undefined, TOOLS); + assert.equal(bodies[0].reasoning_effort, 'high', 'GPT-5.x keeps its effort with tools'); +}); + +test('openai: test() uses max_completion_tokens for reasoning models', async () => { + let bodies = stubFetch([200, OPENAI_OK]); + await openai('gpt-6-astra').test(); + assert.equal('max_tokens' in bodies[0], false); + assert.ok(bodies[0].max_completion_tokens > 0); + + bodies = stubFetch([200, OPENAI_OK]); + await openai('deepseek-v4-pro', { baseURL: 'https://api.deepseek.com' }).test(); + assert.equal(bodies[0].max_tokens, 1); +}); + +test('isReasoningParamError: GPT-6 tools + reasoning_effort rejection is recognised', () => { + assert.equal(isReasoningParamError(JSON.stringify({ error: { message: + "Function tools with reasoning_effort are not supported for gpt-6-astra in /v1/chat/completions. To use function tools, use /v1/responses or set reasoning_effort to 'none'." } })), true); + assert.equal(isReasoningParamError('{"error":{"message":"Incorrect API key provided"}}'), false); +}); + +test('truncateMessages: an overhead larger than the budget keeps the last whole block', async () => { + const { truncateMessages } = await import('../dist/main/main/llm/context-window.js'); + const history = [ + { role: 'user', content: 'make a plate' }, + { role: 'assistant', content: '', toolCalls: [{ id: 'c1', name: 'new_part', parameters: {} }] }, + { role: 'tool', toolCallId: 'c1', content: 'ok' }, + ]; + const kept = truncateMessages(history, 'x '.repeat(200_000), 'claude-opus-5-5', 8192); + assert.notEqual(kept[0]?.role, 'tool', 'a lone tool result is a 400'); + assert.equal(kept[0]?.role, 'assistant'); + assert.equal(kept.length, 2); +}); + +test('llmFetch: connectMs 0 lets a slow non-streaming answer finish', async () => { + const { llmFetch } = await import('../dist/main/main/llm/net.js'); + let calls = 0; + globalThis.fetch = (_u, init) => new Promise((resolve, reject) => { + calls++; + const t = setTimeout(() => resolve(new Response('{}', { status: 200 })), 80); + init.signal?.addEventListener('abort', () => { clearTimeout(t); reject(init.signal.reason); }); + }); + await assert.rejects(llmFetch('https://x.test', {}, 0, { connectMs: 20 }), /connect timeout/); + calls = 0; + const res = await llmFetch('https://x.test', {}, 2, { connectMs: 0 }); + assert.equal(res.status, 200); + assert.equal(calls, 1, 'a finished-but-slow request must not be re-sent'); +}); + +test('anthropic: Sonnet 4.6 keeps temperature when thinking is switched off', async () => { + const bodies = stubFetch([200, CLAUDE_OK]); + await claude('claude-sonnet-4-6', { reasoningLevel: 'off' }).chat(MSGS); + assert.deepEqual(bodies[0].thinking, { type: 'disabled' }); + assert.equal(bodies[0].temperature, 0.3); +}); diff --git a/tests/sidecar-client.test.mjs b/tests/sidecar-client.test.mjs new file mode 100644 index 0000000..28ab96a --- /dev/null +++ b/tests/sidecar-client.test.mjs @@ -0,0 +1,122 @@ +// tests/sidecar-client.test.mjs +// +// P131: the Node sidecar client against the REAL sidecar server (sidecar/sw_agent/server.py), +// with a few fake tools registered by a throwaway bootstrap — no SolidWorks needed. +// - requests go out one at a time; a queued call's budget starts when it is sent +// - a failed call is never re-run by the same-op_id TIMEOUT follow-up +// - heartbeats cannot stretch a call past its absolute cap; onStillRunning fires on them +// - an aborted signal returns CANCELLED at once +// - after an idle gap, back-to-back requests are all answered (P129 stdin-watchdog stall) +// Skipped when no python3 is on PATH. + +import { test, before, after } from 'node:test'; +import assert from 'node:assert/strict'; +import { createRequire } from 'node:module'; +import { spawnSync } from 'node:child_process'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const require = createRequire(import.meta.url); +const Module = require('module'); +const origResolve = Module._resolveFilename; +Module._resolveFilename = function (req, ...rest) { + if (req === 'electron') return 'electron-stub'; + return origResolve.call(this, req, ...rest); +}; +const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'mw-sidecar-')); +require.cache['electron-stub'] = { id: 'electron-stub', filename: 'electron-stub', loaded: true, + exports: { app: { requestSingleInstanceLock: () => true, on() {}, getPath: () => tmp } } }; +const { SWSidecar } = require('../dist/main/main/com/sw-sidecar.js'); + +const PY = ['python3', 'python'].find((p) => spawnSync(p, ['--version']).status === 0); +const SIDECAR = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', 'sidecar'); + +fs.writeFileSync(path.join(tmp, '_bootstrap.py'), ` +import os, sys, time +os.environ["SW_AGENT_SESSION_LOG"] = "0" +sys.path.insert(0, ${JSON.stringify(SIDECAR)}) +from sw_agent import server +from sw_agent.registry import tool +from sw_agent.bridge import SWError + +server.HEARTBEAT_S = 0.1 +RUNS = {} + +def _bump(k): + RUNS[k] = RUNS.get(k, 0) + 1 + return RUNS[k] + +@tool("t_sleep", "sleeps", params={"secs": {"type": "number"}}, category="query", internal=True) +def t_sleep(ctx, secs=0.1): + time.sleep(secs) + return {"slept": secs} + +@tool("t_fail", "fails after working", params={"secs": {"type": "number"}}, category="query", internal=True) +def t_fail(ctx, secs=0.1): + n = _bump("fail") + time.sleep(secs) + raise SWError(f"half-built, then failed (run {n})") + +server.serve() +`); + +let sc; +before(async () => { + if (!PY) return; + sc = new SWSidecar({ pythonPath: PY, cwd: tmp }); + await sc.start(); +}); +after(() => { sc?.stop(); fs.rmSync(tmp, { recursive: true, force: true }); }); + +const opts = { skip: !PY && 'python3 not available' }; + +test('sidecar: a queued call is not timed out while it waits its turn', opts, async () => { + const slow = sc.call('t_sleep', { secs: 0.8 }, { timeoutMs: 5000 }); + const queued = sc.call('t_sleep', { secs: 0.05 }, { timeoutMs: 400 }); // < slow's run time + const [a, b] = await Promise.all([slow, queued]); + assert.equal(a.ok, true); + assert.equal(b.ok, true, `queued call failed: ${b.code} ${b.error}`); +}); + +test('sidecar: the TIMEOUT follow-up never re-runs a failed call', opts, async () => { + const r = await sc.call('t_fail', { secs: 0.6 }, { timeoutMs: 300, opId: 'fail-once' }); + assert.equal(r.ok, false); + assert.match(String(r.error), /run 1\)/, 'the follow-up must return the cached failure of run 1'); + const again = await sc.call('t_fail', { secs: 0 }, { timeoutMs: 2000, opId: 'fail-once' }); + assert.match(String(again.error), /run 1\)/); +}); + +test('sidecar: heartbeats cannot stretch a call past its cap; progress is reported', opts, async () => { + let beats = 0; + const t0 = Date.now(); + const r = await sc.call('t_sleep', { secs: 3 }, { timeoutMs: 300, onStillRunning: () => beats++ }); + const took = Date.now() - t0; + assert.equal(r.code, 'TIMEOUT'); + assert.ok(took < 2500, `took ${took} ms — the 900 ms cap (+300 ms follow-up) was not enforced`); + assert.ok(beats >= 3, `onStillRunning fired ${beats}×`); + assert.ok(r.durationMs >= took - 50, 'durationMs covers the follow-up too'); + await sc.call('t_sleep', { secs: 0 }, { timeoutMs: 5000 }); // drain: wait for the 3 s job +}); + +test('sidecar: an aborted signal returns CANCELLED at once', opts, async () => { + const ac = new AbortController(); + setTimeout(() => ac.abort(), 100); + const t0 = Date.now(); + const r = await sc.call('t_sleep', { secs: 1 }, { timeoutMs: 5000, signal: ac.signal }); + assert.equal(r.code, 'CANCELLED'); + assert.ok(Date.now() - t0 < 600); + const p = await sc.ping(); // queued behind the abandoned job, then answered + assert.equal(p.ok, true); +}); + +test('sidecar: back-to-back requests after an idle gap are all answered', opts, async () => { + await new Promise((r) => setTimeout(r, 2500)); // P129's stdin watchdog peeked every 2 s + for (let i = 0; i < 3; i++) { + const t0 = Date.now(); + const p = await sc.ping(); + assert.equal(p.ok, true, `ping ${i}: ${p.code}`); + assert.ok(Date.now() - t0 < 1000, `ping ${i} took ${Date.now() - t0} ms`); + } +});