diff --git a/CHANGELOG.md b/CHANGELOG.md index 6590edb..1385f34 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,32 @@ All notable changes follow Keep a Changelog. Versions follow Semantic Versioning ## [Unreleased] +### Added + +- Provider-backed `run` and `chat` CLI commands that compose the existing Agent runtime, + OpenAI-compatible or Anthropic adapters, bounded Workspace, and governed built-in tools. +- SiliconFlow configuration through `provider`, `model`, `base_url`, and the existing + `MINI_CODE_AGENT_OPENAI_API_KEY` secret environment variable. +- Rich terminal lifecycle output and explicit action previews for file writes and local argv + commands. + +### Security + +- Read-only tools remain allowed by default. Writes and CLI-enabled command execution require an + interactive approval; non-interactive mode denies both without prompting. +- CLI output uses normalized public errors and never renders API key values. Live provider calls + remain outside CI. +- Approval previews render model-controlled text literally and quote argv using platform-specific + display rules so Rich markup or whitespace cannot obscure argument boundaries. + +### Verification + +- Local uv-managed Python 3.13.14 passed 1201 tests with 13 Windows privilege/platform skips and + 88.56% branch-aware package coverage. Ruff format/check and strict Pyright passed. +- MockTransport verified the SiliconFlow-compatible + `https://api.siliconflow.cn/v1/chat/completions` request path and bearer header without a live + credential. A live SiliconFlow request was not run. + ## [0.16.0-alpha.0] - 2026-07-02 ### Added diff --git a/README.md b/README.md index b473e35..354fbe2 100644 --- a/README.md +++ b/README.md @@ -2,14 +2,15 @@ A framework-light, provider-neutral coding agent built from first principles. -> Status: pre-alpha. M6b provides a provider-neutral Agent Core, Anthropic/OpenAI-compatible +> Status: pre-alpha. M7 provides a provider-neutral Agent Core, Anthropic/OpenAI-compatible > adapters, a schema-validating Tool Registry, a cross-platform Workspace boundary, bounded > Read/Search, conflict-aware Write/Edit, policy-governed argv command execution, and deterministic > context admission, hardened read-only Git evidence, governed Pytest diagnostics, versioned SQLite > Session/Trace persistence, fail-closed Checkpoint/Resume, and a host-controlled bounded Repair > loop, provenance-aware lazy Skills, deterministic host-registered Tool Hooks, and host-pinned > local MCP stdio Tools, bounded host-profiled read-only analysis Subagents, and host-managed -> Worktree implementation candidates with separately approved adoption. OS sandboxing, +> Worktree implementation candidates with separately approved adoption, plus provider-backed +> `run` and `chat` terminal commands with governed action previews. OS sandboxing, > shell-string execution, project-provided executable Hooks, automatic Repair resume, remote > HTTP/OAuth MCP, automatic commit/merge/push, and live-provider CI are not implemented. @@ -46,6 +47,41 @@ Default config paths follow the operating system conventions provided by Platfor Secrets are accepted from environment variables but are never printed by `doctor`. See `config.example.toml` for supported inputs. +## Run with SiliconFlow + +Create a local config file from the example and set the API key only in the current shell: + +```powershell +Copy-Item .\config.example.toml .\config.toml +$env:MINI_CODE_AGENT_OPENAI_API_KEY = "your-siliconflow-api-key" +``` + +Use the exact model identifier available in your SiliconFlow account. The example uses +`Pro/zai-org/GLM-4.7` with the OpenAI-compatible endpoint +`https://api.siliconflow.cn/v1`. + +Run one coding task against the current workspace: + +```powershell +mini-code-agent run "Inspect this project and summarize its architecture." ` + --config .\config.toml ` + --workspace . +``` + +Start an interactive task loop: + +```powershell +mini-code-agent chat --config .\config.toml --workspace . +``` + +Each `chat` prompt starts an independent bounded Agent run against the same workspace; durable +conversation memory is not implied. Read-only tools run automatically. File writes and local argv +commands display an action preview and require explicit confirmation. Use `--non-interactive` with +`run` to deny writes and commands instead of prompting. + +These commands call the configured live model API and consume provider quota. CI uses mocked HTTP +transports and never requires a real API key. + ## Provider Adapters Both adapters implement the same `ModelProvider` protocol: @@ -290,7 +326,9 @@ claim token/latency improvements. See - Product design: `docs/superpowers/specs/2026-06-29-mini-code-agent-design.md` - Learning map: `docs/learning/knowledge-map.md` - Learning evidence: `docs/learning/progress.md` +- M7 CLI learning notes: `docs/learning/m7-cli-provider-runtime.md` - Resume evidence: `docs/resume/project-profile.md` +- M7 resume and interview profile: `docs/resume/m7-cli-project-profile.md` - Agent Core: `docs/architecture/agent-core.md` - Provider adapters: `docs/architecture/provider-adapters.md` - Read-only tools: `docs/architecture/readonly-tools.md` diff --git a/config.example.toml b/config.example.toml index a5d4fdf..56cc62d 100644 --- a/config.example.toml +++ b/config.example.toml @@ -2,3 +2,9 @@ log_level = "info" data_dir = ".mini-code-agent" trace_enabled = true +provider = "openai_compatible" +model = "Pro/zai-org/GLM-4.7" +base_url = "https://api.siliconflow.cn/v1" + +# Keep API keys out of this file. For SiliconFlow, set: +# MINI_CODE_AGENT_OPENAI_API_KEY diff --git a/docs/learning/m7-cli-provider-runtime.md b/docs/learning/m7-cli-provider-runtime.md new file mode 100644 index 0000000..eeeec61 --- /dev/null +++ b/docs/learning/m7-cli-provider-runtime.md @@ -0,0 +1,119 @@ +# M7 学习笔记:把 Agent SDK 组装成可使用的 CLI 产品 + +## 1. 这一阶段解决什么问题 + +M6b 之前,项目已经有 Agent Loop、Provider、Tool、Policy、Workspace 等核心模块,但用户 +需要自己写 Python 代码才能把它们组装起来。M7 增加应用组合根和 `run/chat` 命令,让真实 +模型、工作区和受治理工具形成一个可以直接体验的闭环。 + +关键链路是: + +```text +CLI 参数/配置 + -> Provider Factory + -> Workspace + Tool Registry + -> Policy + Approval + -> AgentRuntime + -> OpenAI-compatible API + -> ToolCall / ToolResult + -> 最终回答 +``` + +## 2. 需要掌握的知识点 + +### 2.1 Composition Root + +`application.py` 是组合根。它不实现新的 Agent 算法,只负责创建并连接已有模块: + +- 根据 `AppSettings` 创建 Provider; +- 根据工作区创建文件、Git 和命令工具; +- 用 `GovernedToolExecutor` 包装有副作用的工具; +- 创建 `AgentRuntime` 并执行任务; +- 在 `finally` 中关闭 Provider HTTP Client。 + +这和 Spring Boot 的配置类/Bean 装配类似。区别是 Python 项目没有依赖注入容器,依赖关系 +由普通函数显式构造,因此调用顺序、所有权和测试替身更加直接。 + +### 2.2 OpenAI-compatible API + +硅基流动使用 OpenAI-compatible Chat Completions 协议。项目只需要配置: + +```text +provider = openai_compatible +model = 账户中实际可用的模型 ID +base_url = https://api.siliconflow.cn/v1 +``` + +`OpenAICompatibleProvider` 会在 base URL 后追加 `chat/completions`。模型返回的 +`tool_calls` 被规范化为项目内部 `ToolCall`,工具执行结果再转换回兼容协议消息。 + +### 2.3 配置与密钥边界 + +配置优先级仍然是: + +```text +默认值 < TOML < MINI_CODE_AGENT_* 环境变量 < 显式 overrides +``` + +`provider/model/base_url` 可以进入 TOML;API Key 只建议使用环境变量。`SecretStr` 和 +`safe_dict()` 防止 `doctor`、错误日志或诊断输出直接显示密钥。 + +### 2.4 同步 CLI 与异步 Agent + +Typer 命令是同步函数,而 Provider、Tool 和 Agent Runtime 是异步接口。CLI 使用 +`asyncio.run()` 建立每次任务的 event loop。这适合普通终端进程,但不能直接复制到已经拥有 +event loop 的 Jupyter、FastAPI 请求处理或异步测试中;这些场景应直接 `await run_task()`。 + +### 2.5 Capability 与 Approval + +CLI 注册工具不代表模型自动获得执行授权: + +- Read/Search/Git status/diff 是只读能力,默认允许; +- Write/Edit 默认进入 `ASK`; +- Execute 默认是 `DENY`,M7 只为 `run_command` 增加明确的 `ASK` 规则; +- `--non-interactive` 不弹审批,而是拒绝所有 `ASK` 操作。 + +审批器展示工具名、风险、理由、资源、argv 和 bounded diff。模型只能提出动作,不能替用户 +批准动作。 + +### 2.6 资源所有权 + +Provider 内部创建的 `httpx.AsyncClient` 必须关闭。`run_task()` 在成功、Provider 错误和 +Runtime 失败时都进入 `finally`。测试通过可关闭的 `ScriptedProvider` 验证这一点。 + +## 3. Java / Flink / Spark SQL 经验映射 + +| 现有经验 | M7 对应概念 | 关键差异 | +|---|---|---| +| Spring `@Configuration` | `application.py` composition root | 依赖由普通函数显式创建,没有容器生命周期 | +| Spring `@ConfigurationProperties` | Pydantic `AppSettings` | 字段校验与环境变量合并由 Pydantic Settings 完成 | +| Feign/WebClient | `OpenAICompatibleProvider` + `httpx` | 响应不仅是 DTO,还可能驱动 ToolCall 状态机 | +| Java `try/finally` / `AutoCloseable` | Provider `aclose()` in `finally` | HTTP Client 是异步资源,需要 `await` | +| RBAC/接口鉴权 | Tool Policy allow/ask/deny | 授权对象是一次具体的模型动作,不是用户页面权限 | +| Flink operator graph | Provider -> Runtime -> Tool pipeline | Agent 路径由模型响应动态决定,不是预先固定 DAG | +| SQL dry-run / execution plan | `ActionPreview` | 预览只用于人工决策,不能替代执行前重验证 | + +## 4. 建议练习 + +1. 用 `httpx.MockTransport` 抓取请求,确认 SiliconFlow URL、Authorization 和 model 字段。 +2. 让模拟 Provider 先调用 `read_file`,再返回最终答案,画出完整消息序列。 +3. 让模拟 Provider 请求 `write_file`,分别验证批准、拒绝和 non-interactive 三条路径。 +4. 修改模型 ID 或 base URL,观察错误发生在配置期、网络期还是协议解析期。 +5. 阅读 `tests/unit/test_application.py`,解释为什么真实 API 不应进入默认 CI。 + +## 5. 当前边界 + +- `chat` 是同一工作区上的交互任务循环,每条输入是独立 Agent run,不是持久对话。 +- Runtime 当前使用非流式 `complete()`,终端展示生命周期事件,但不逐 Token 输出。 +- CLI 尚未组合 SQLite Trace/Checkpoint、Skills、MCP、Subagent 和 Worktree candidate。 +- 真实 SiliconFlow 调用需要用户本地 API Key,并会消耗账户额度。 +- 命令治理和 Workspace boundary 不是 OS sandbox。 + +## 6. 当前验证证据 + +- uv 管理的 Python 3.13.14:1201 passed、13 个 Windows 权限/平台条件 skip。 +- branch-aware package coverage:88.56%,高于 85% 门槛。 +- Ruff format/check:通过。 +- strict Pyright:0 errors。 +- `httpx.MockTransport`:验证 SiliconFlow-compatible URL 和 Bearer Header。 +- 真实 SiliconFlow API:未执行,因为验证环境没有配置用户 API Key。 diff --git a/docs/resume/m7-cli-project-profile.md b/docs/resume/m7-cli-project-profile.md new file mode 100644 index 0000000..d0e92fa --- /dev/null +++ b/docs/resume/m7-cli-project-profile.md @@ -0,0 +1,105 @@ +# Mini CodeAgent M7 简历与面试说明 + +## 项目介绍 + +Mini CodeAgent 是一个从零实现、Framework-light、Provider-neutral 的 Python Coding Agent。 +项目将大模型 Tool Calling、受限工作区、工具权限、人工审批、上下文预算、Trace/Checkpoint、 +MCP、Subagent 和 Worktree candidate 拆成可测试模块。M7 在此基础上增加 Provider-backed +CLI,使用户可以通过 SiliconFlow 等 OpenAI-compatible 服务直接执行代码分析和修改任务。 + +## 技术栈 + +- Python 3.12/3.13、asyncio、强类型 Protocol +- Pydantic、Pydantic Settings +- Typer、Rich +- httpx、OpenAI-compatible Chat Completions、Anthropic Messages +- JSON Schema Draft 2020-12 +- Pytest、pytest-asyncio、httpx MockTransport +- Ruff、Pyright、GitHub Actions + +## 面试介绍版本 + +“我做了一个 Python Mini Coding Agent。核心不是简单调用一次大模型,而是实现 +Model -> ToolCall -> ToolResult -> Model 的有界执行循环,并把文件、Git 和本地命令封装成 +受治理工具。M7 增加了应用组合根和 Typer CLI,可以切换 OpenAI-compatible 或 Anthropic +Provider,也可以直接接硅基流动。读操作默认允许,写文件和执行命令必须先展示资源、argv 或 +diff,再由用户批准;非交互模式全部 fail closed。真实密钥只从环境变量读取,测试使用 +MockTransport 和 ScriptedProvider,不消耗线上额度。” + +## 项目亮点 + +### 1. Provider-neutral 的真实模型接入 + +**为什么使用:** 避免 Agent 核心绑定某一家模型厂商,也便于利用不同平台的额度和模型能力。 + +**技术实现:** 通过统一 `ModelProvider` 协议隔离内部消息模型与厂商 wire protocol; +`build_provider()` 根据强类型配置创建 OpenAI-compatible 或 Anthropic Adapter。硅基流动只需 +配置 model、base URL 和环境变量 API Key。 + +**实现功能:** 同一套 Agent Runtime 可以调用不同 Provider,并支持 Tool Calling。 + +**解决问题:** 消除业务循环中的厂商条件分支,降低更换模型服务时的修改范围。 + +**证据:** `src/mini_code_agent/application.py`、 +`tests/unit/test_application.py::test_openai_compatible_provider_uses_siliconflow_endpoint`。 + +### 2. 显式 Composition Root + +**为什么使用:** SDK 模块齐全不等于产品可运行;必须有一个地方明确管理依赖、资源和安全策略。 + +**技术实现:** `application.py` 统一创建 Provider、Workspace、Tool Registry、Policy、 +Approval 和 AgentRuntime,并在 `finally` 中关闭 Provider Client。 + +**实现功能:** `run` 和 `chat` 不需要重复装配代码,测试可以注入 ScriptedProvider。 + +**解决问题:** 避免 CLI、Web 或测试各自复制装配逻辑,减少策略不一致和资源泄漏。 + +**证据:** `run_task()` 及成功/失败关闭资源测试。 + +### 3. 模型不可自授权的工具治理 + +**为什么使用:** Coding Agent 会修改文件和执行命令,模型输出不能直接等价为用户授权。 + +**技术实现:** Tool Definition 标记 side effect;Policy 产生 allow/ask/deny;写入沿用默认 +ASK,命令执行从默认 DENY 提升为 CLI 明确 ASK;`TerminalApprovalHandler` 展示 ActionPreview +后才返回决定。 + +**实现功能:** 只读分析自动执行,写文件和 argv 命令逐次审批,non-interactive 模式自动拒绝。 + +**解决问题:** 阻止模型静默落盘或执行本地进程,并为用户提供可判断的资源和 diff 信息。 + +**证据:** `build_tool_executor()`、`TerminalApprovalHandler` 及审批测试。 + +### 4. 密钥安全与可测试的线上协议 + +**为什么使用:** API Key 泄漏和 CI 消耗真实 Token 都是 Agent 项目的常见工程风险。 + +**技术实现:** Key 使用 `SecretStr` 和环境变量;诊断只输出 configured 布尔值;HTTP 测试使用 +`httpx.MockTransport` 验证真实 URL/Header/Response parsing,不连接公网。 + +**实现功能:** 能验证 SiliconFlow 接口兼容性,同时保持默认测试确定性。 + +**解决问题:** 防止密钥进入仓库、日志和测试报告,也避免 CI 受网络、额度和模型漂移影响。 + +**证据:** `AppSettings.safe_dict()`、配置测试和 SiliconFlow endpoint 测试。 + +### 5. 稳定的 CLI 错误与退出码 + +**为什么使用:** Agent 可能因配置、鉴权、限流或工具边界停止,脚本和用户需要区分失败类型。 + +**技术实现:** 配置/组合错误返回退出码 2;Agent 未完成返回 1;完成返回 0;终端只显示 bounded +public error 和运行摘要。 + +**实现功能:** 既适合人工使用,也可被 PowerShell、CI 或其他进程可靠调用。 + +**解决问题:** 避免所有错误都变成 Traceback 或模糊的非零状态。 + +**证据:** `tests/cli/test_cli.py` 中 run 成功、配置错误和 Provider 错误测试。 + +## 诚实边界 + +- 不描述为 Claude Code 的完整替代品;当前没有全屏 TUI 或 Web UI。 +- 不声称 `chat` 已有跨 run 的持久对话记忆。 +- 不声称默认 CI 已验证真实 SiliconFlow 服务;真实 smoke 需要本地凭证。 +- 不把 Workspace/Policy 描述为 OS sandbox。 +- 不编造 Token、准确率、开发效率等百分比提升;没有 benchmark 就不写量化结果。 diff --git a/docs/superpowers/plans/2026-07-02-m7-cli-provider-runtime.md b/docs/superpowers/plans/2026-07-02-m7-cli-provider-runtime.md new file mode 100644 index 0000000..6fb7518 --- /dev/null +++ b/docs/superpowers/plans/2026-07-02-m7-cli-provider-runtime.md @@ -0,0 +1,216 @@ +# M7 CLI Provider Runtime Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [x]`) syntax for tracking. + +**Goal:** Add provider-backed `run` and `chat` commands that make the existing governed Agent runtime usable with SiliconFlow. + +**Architecture:** Extend immutable settings with provider connection metadata, then add a small application composition root and terminal adapters. Keep Typer commands thin, preserve secure policy defaults, and inject fakes at module boundaries for deterministic tests. + +**Tech Stack:** Python 3.12+, Pydantic Settings, Typer, Rich, httpx, pytest, pytest-asyncio + +--- + +### Task 1: Provider Runtime Configuration + +**Files:** +- Modify: `src/mini_code_agent/config.py` +- Modify: `tests/unit/test_config.py` + +- [x] **Step 1: Write failing configuration tests** + +Add tests that load `provider`, `model`, and `base_url` from TOML and override them with +`MINI_CODE_AGENT_PROVIDER`, `MINI_CODE_AGENT_MODEL`, and `MINI_CODE_AGENT_BASE_URL`. Assert that +`safe_dict` exposes only non-secret provider metadata. + +- [x] **Step 2: Verify the tests fail** + +Run: `python -m pytest tests/unit/test_config.py -q` + +Expected: failure because `ProviderName` and the new settings fields do not exist. + +- [x] **Step 3: Implement the minimal settings fields** + +Add `ProviderName`, optional `model`, optional `base_url`, matching environment fields, bounded +validation, and secret-safe rendering. + +- [x] **Step 4: Verify the tests pass** + +Run: `python -m pytest tests/unit/test_config.py -q` + +Expected: all configuration tests pass. + +### Task 2: Application Composition Root + +**Files:** +- Create: `src/mini_code_agent/application.py` +- Create: `tests/unit/test_application.py` + +- [x] **Step 1: Write failing provider factory tests** + +Specify `ApplicationConfigurationError`, `build_provider`, and `build_tool_executor`. Cover +SiliconFlow URL/model propagation through `httpx.MockTransport`, missing provider credentials, +the Anthropic selection, registered tool names, and the execute ask policy. + +- [x] **Step 2: Verify the tests fail** + +Run: `python -m pytest tests/unit/test_application.py -q` + +Expected: import failure because `application.py` does not exist. + +- [x] **Step 3: Implement provider and tool composition** + +Create providers from `AppSettings`, build the bounded workspace tool registry, add an explicit +`ASK` policy rule for `run_command`, and return a governed executor. + +- [x] **Step 4: Add and verify the async task runner tests** + +Use `ScriptedProvider` and an injected provider factory to prove success, failure propagation, and +provider closure without network access. + +- [x] **Step 5: Run the application tests** + +Run: `python -m pytest tests/unit/test_application.py -q` + +Expected: all application tests pass. + +### Task 3: Terminal Adapters + +**Files:** +- Create: `src/mini_code_agent/terminal.py` +- Create: `tests/unit/test_terminal.py` + +- [x] **Step 1: Write failing approval and event rendering tests** + +Specify a callback-injected `TerminalApprovalHandler` and `TerminalEventSink`. Assert that action +summaries, resources, argv, and diffs are shown, the callback decision is returned, and event +output contains no prompt or tool payload. + +- [x] **Step 2: Verify the tests fail** + +Run: `python -m pytest tests/unit/test_terminal.py -q` + +Expected: import failure because `terminal.py` does not exist. + +- [x] **Step 3: Implement bounded Rich terminal rendering** + +Render approval previews and compact lifecycle status. Keep all model-controlled values bounded +by their existing Pydantic models. + +- [x] **Step 4: Verify the tests pass** + +Run: `python -m pytest tests/unit/test_terminal.py -q` + +Expected: all terminal tests pass. + +### Task 4: Run and Chat Commands + +**Files:** +- Modify: `src/mini_code_agent/cli.py` +- Modify: `tests/cli/test_cli.py` + +- [x] **Step 1: Write failing `run` command tests** + +Patch the async application boundary, invoke Typer, and assert final text, summary, workspace +forwarding, non-interactive mode, configuration exit code two, and Agent failure exit code one. + +- [x] **Step 2: Verify the run tests fail** + +Run: `python -m pytest tests/cli/test_cli.py -q` + +Expected: failure because the `run` command does not exist. + +- [x] **Step 3: Implement `run`** + +Load settings, configure logging, construct terminal adapters, execute the coroutine through +`asyncio.run`, render the result, and map errors to exit codes. + +- [x] **Step 4: Write failing `chat` command tests** + +Patch prompts with two tasks followed by `/exit`; assert two application calls and no call for an +empty prompt. + +- [x] **Step 5: Verify the chat tests fail** + +Run: `python -m pytest tests/cli/test_cli.py -q` + +Expected: failure because the `chat` command does not exist. + +- [x] **Step 6: Implement `chat`** + +Reuse the same settings, workspace, system prompt, and rendering helpers for each independent +bounded run. Handle `/exit`, `/quit`, EOF, and Ctrl+C without a traceback. + +- [x] **Step 7: Verify all CLI tests pass** + +Run: `python -m pytest tests/cli/test_cli.py -q` + +Expected: all CLI tests pass. + +### Task 5: Documentation and Release Metadata + +**Files:** +- Modify: `config.example.toml` +- Modify: `README.md` +- Modify: `CHANGELOG.md` +- Modify: `pyproject.toml` +- Modify: `src/mini_code_agent/__init__.py` +- Modify: `tests/cli/test_cli.py` + +- [x] **Step 1: Document SiliconFlow configuration** + +Add PowerShell environment setup, the exact base URL, model placeholder, `run`, `chat`, approval, +and live-smoke instructions. State that CI does not use live credentials. + +- [x] **Step 2: Bump the alpha version** + +Set the package and module version to `0.17.0a0`, update the CLI assertion, and add a changelog +entry. + +- [x] **Step 3: Run focused checks** + +Run: `python -m pytest tests/unit/test_config.py tests/unit/test_application.py tests/unit/test_terminal.py tests/cli/test_cli.py -q` + +Expected: all focused tests pass. + +### Task 6: Full Verification + +**Files:** +- Verify only + +- [x] **Step 1: Format and lint** + +Run: `python -m ruff format --check .` + +Expected: exit zero. + +Run: `python -m ruff check .` + +Expected: exit zero. + +- [x] **Step 2: Type-check** + +Run: `python -m pyright` + +Expected: zero errors. + +- [x] **Step 3: Run the full test suite** + +Prepend `.venv/Scripts` to `PATH`, then run: `python -m pytest -q` + +Expected: all runnable tests pass; Windows privilege-dependent symlink tests may skip. + +- [x] **Step 4: Run CLI smoke checks** + +Run: `mini-code-agent --version` + +Expected: `0.17.0a0`. + +Run: `mini-code-agent run "Inspect this project" --non-interactive` without a model/key. + +Expected: exit two with an actionable configuration error and no traceback or secret. + +- [x] **Step 5: Review the diff** + +Run: `git diff --check` and `git status --short`. + +Expected: no whitespace errors and only M7 files are changed. diff --git a/docs/superpowers/specs/2026-07-02-m7-cli-provider-runtime-design.md b/docs/superpowers/specs/2026-07-02-m7-cli-provider-runtime-design.md new file mode 100644 index 0000000..574c404 --- /dev/null +++ b/docs/superpowers/specs/2026-07-02-m7-cli-provider-runtime-design.md @@ -0,0 +1,103 @@ +# M7 CLI Provider Runtime Design + +## Goal + +Turn the existing Agent SDK into a usable terminal product that can execute one task or an +interactive sequence of tasks against an OpenAI-compatible provider such as SiliconFlow. + +## Scope + +M7 adds: + +- provider, model, and base URL configuration; +- `mini-code-agent run TASK`; +- `mini-code-agent chat`; +- a composition root that connects providers, the workspace, governed tools, and `AgentRuntime`; +- terminal approval previews for writes and command execution; +- deterministic tests that never call a live model API; +- SiliconFlow setup and usage documentation. + +M7 does not add a Web UI, full-screen TUI, streaming model output, durable chat continuity, or +automatic command approval. Each `chat` prompt starts a new bounded Agent run while sharing the +same workspace. + +## Architecture + +`config.py` remains the only source of configuration precedence. It adds a provider selector, +model identifier, and optional base URL. Secrets continue to come from the existing environment +key fields and are never rendered by `safe_dict`. + +`application.py` is the composition root. It validates runtime-specific settings, creates the +selected provider, constructs a `WorkspaceBoundary`, registers built-in tools, wraps them in +`GovernedToolExecutor`, executes `AgentRuntime`, and closes the provider-owned HTTP client. + +`terminal.py` owns terminal-specific behavior. Its approval handler renders the model's proposed +action, affected resources, argv, and bounded diff before obtaining a yes/no decision. Its event +sink renders bounded progress without exposing prompts, arguments, tool results, or secrets. + +`cli.py` remains a thin Typer adapter. It loads configuration, selects interactive or +non-interactive policy mode, calls the application service, renders the final result, and maps +configuration or runtime failures to stable exit codes. + +## Provider Configuration + +The supported provider values are: + +- `openai_compatible`: uses `openai_api_key`; defaults to `https://api.openai.com/v1`. +- `anthropic`: uses `anthropic_api_key`; defaults to `https://api.anthropic.com`. + +For SiliconFlow: + +```toml +[mini_code_agent] +provider = "openai_compatible" +model = "MODEL_FROM_SILICONFLOW_ACCOUNT" +base_url = "https://api.siliconflow.cn/v1" +``` + +The API key is supplied as `MINI_CODE_AGENT_OPENAI_API_KEY`. A missing model, missing matching +key, or invalid provider construction is reported before any Agent work begins. + +## Tool Governance + +The CLI registers `read_file`, `search_text`, `write_file`, `edit_file`, `git_status`, +`git_diff`, and `run_command`. + +- Read-only tools use the existing default allow rule. +- Writes use the existing default ask rule. +- `run_command` receives an explicit ask rule because execute is denied by default. +- Interactive sessions call the terminal approval handler. +- Non-interactive sessions deny all ask decisions without prompting. + +The workspace boundary and argv-only command runner remain the enforcement points. M7 does not +claim OS sandboxing. + +## Command Behavior + +`run TASK` executes one bounded Agent run. It prints the final model text, then a compact run +summary. A completed run exits zero; configuration errors exit two; incomplete Agent runs exit +one. + +`chat` prompts until `/exit`, `/quit`, EOF, or Ctrl+C. Every non-empty prompt invokes the same +single-run application flow and shares filesystem changes through the selected workspace. It does +not silently claim conversational memory. + +## Error Handling + +Public errors are bounded and omit secret values. Provider failures use the existing normalized +`AgentResult.error`. Configuration errors are caught at the CLI boundary. Provider clients are +closed in `finally`, including interrupted and failed runs. + +## Testing + +Tests follow red-green-refactor and cover: + +- configuration precedence and secret-safe diagnostics; +- provider selection, SiliconFlow base URL propagation, and missing-key errors; +- built-in tool and policy composition; +- approval rendering and decisions; +- `run` success/failure exit behavior; +- `chat` prompt loop and exit commands; +- a mocked OpenAI-compatible end-to-end CLI flow without network access. + +Live SiliconFlow verification is an explicit local smoke test and is not part of CI. diff --git a/pyproject.toml b/pyproject.toml index 56aa6b2..2bf93e1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "mini-code-agent" -version = "0.16.0a0" +version = "0.17.0a0" description = "A framework-light, provider-neutral, enterprise-grade mini code agent." readme = "README.md" requires-python = ">=3.12,<3.14" diff --git a/src/mini_code_agent/application.py b/src/mini_code_agent/application.py new file mode 100644 index 0000000..b696a86 --- /dev/null +++ b/src/mini_code_agent/application.py @@ -0,0 +1,165 @@ +from __future__ import annotations + +from collections.abc import Awaitable, Callable +from pathlib import Path +from typing import Protocol, cast + +import httpx + +from mini_code_agent.agent.events import EventSink +from mini_code_agent.agent.models import AgentResult +from mini_code_agent.agent.runtime import AgentRuntime +from mini_code_agent.command.runner import CommandRunner +from mini_code_agent.config import AppSettings, ProviderName +from mini_code_agent.git.client import GitClient +from mini_code_agent.policy.approval import ApprovalHandler +from mini_code_agent.policy.engine import PolicyEngine +from mini_code_agent.policy.executor import GovernedToolExecutor +from mini_code_agent.policy.models import ( + PolicyDecision, + PolicyRule, + SessionMode, + TrustSource, +) +from mini_code_agent.providers.anthropic import AnthropicProvider +from mini_code_agent.providers.base import ModelProvider +from mini_code_agent.providers.openai_compatible import OpenAICompatibleProvider +from mini_code_agent.tools.base import SideEffect +from mini_code_agent.tools.edit_file import EditFileTool +from mini_code_agent.tools.git_diff import GitDiffTool +from mini_code_agent.tools.git_status import GitStatusTool +from mini_code_agent.tools.read_file import ReadFileTool +from mini_code_agent.tools.registry import ToolRegistry +from mini_code_agent.tools.run_command import RunCommandTool +from mini_code_agent.tools.search_text import SearchTextTool +from mini_code_agent.tools.write_file import WriteFileTool +from mini_code_agent.workspace.boundary import WorkspaceBoundary + +DEFAULT_SYSTEM_PROMPT = """\ +You are a coding agent working inside a bounded local workspace. +Inspect relevant files before changing them. Use workspace-relative paths. +Explain the completed work concisely and report any verification you could not run. +""" + +type HttpProvider = AnthropicProvider | OpenAICompatibleProvider + + +class ApplicationConfigurationError(ValueError): + """Raised when the CLI runtime cannot be composed from trusted settings.""" + + +class ProviderFactory(Protocol): + def __call__(self, settings: AppSettings) -> ModelProvider: ... + + +def build_provider( + settings: AppSettings, + *, + client: httpx.AsyncClient | None = None, +) -> HttpProvider: + if settings.model is None: + raise ApplicationConfigurationError( + "A model is required. Set MINI_CODE_AGENT_MODEL or model in config.toml." + ) + try: + if settings.provider is ProviderName.OPENAI_COMPATIBLE: + if settings.openai_api_key is None: + raise ApplicationConfigurationError( + "Set MINI_CODE_AGENT_OPENAI_API_KEY for the openai_compatible provider." + ) + return OpenAICompatibleProvider( + api_key=settings.openai_api_key, + model=settings.model, + base_url=settings.base_url or "https://api.openai.com/v1", + client=client, + ) + if settings.anthropic_api_key is None: + raise ApplicationConfigurationError( + "Set MINI_CODE_AGENT_ANTHROPIC_API_KEY for the anthropic provider." + ) + return AnthropicProvider( + api_key=settings.anthropic_api_key, + model=settings.model, + base_url=settings.base_url or "https://api.anthropic.com", + client=client, + ) + except ApplicationConfigurationError: + raise + except ValueError: + raise ApplicationConfigurationError( + "Provider model or base URL configuration is invalid." + ) from None + + +def build_tool_executor( + workspace_root: Path, + *, + approval: ApprovalHandler, + session_mode: SessionMode, +) -> GovernedToolExecutor: + try: + workspace = WorkspaceBoundary(workspace_root) + git = GitClient(workspace.root) + except ValueError: + raise ApplicationConfigurationError( + "Workspace must be an existing local directory." + ) from None + registry = ToolRegistry( + ( + ReadFileTool(workspace), + SearchTextTool(workspace), + WriteFileTool(workspace), + EditFileTool(workspace), + GitStatusTool(git), + GitDiffTool(git), + RunCommandTool(workspace, CommandRunner()), + ) + ) + policy = PolicyEngine( + ( + PolicyRule( + id="cli-ask-execute", + decision=PolicyDecision.ASK, + rationale="Local command execution requires explicit terminal approval.", + tool_glob="run_command", + side_effect=SideEffect.EXECUTE, + ), + ) + ) + return GovernedToolExecutor( + registry, + policy=policy, + approval=approval, + session_mode=session_mode, + trust_source=TrustSource.MODEL, + ) + + +async def run_task( + settings: AppSettings, + *, + workspace: Path, + user_prompt: str, + approval: ApprovalHandler, + session_mode: SessionMode, + system_prompt: str = DEFAULT_SYSTEM_PROMPT, + events: EventSink | None = None, + provider_factory: ProviderFactory | None = None, +) -> AgentResult: + factory = provider_factory or build_provider + provider = factory(settings) + try: + tools = build_tool_executor( + workspace, + approval=approval, + session_mode=session_mode, + ) + return await AgentRuntime(provider, tools, events=events).run( + user_prompt=user_prompt, + system_prompt=system_prompt, + ) + finally: + close_candidate = getattr(provider, "aclose", None) + if close_candidate is not None: + close = cast(Callable[[], Awaitable[None]], close_candidate) + await close() diff --git a/src/mini_code_agent/cli.py b/src/mini_code_agent/cli.py index 365b276..e929d80 100644 --- a/src/mini_code_agent/cli.py +++ b/src/mini_code_agent/cli.py @@ -1,5 +1,6 @@ from __future__ import annotations +import asyncio import json from pathlib import Path from typing import Annotated @@ -7,15 +8,25 @@ import typer from rich.console import Console from rich.table import Table +from rich.text import Text from mini_code_agent import __version__ +from mini_code_agent.agent.models import AgentResult +from mini_code_agent.application import ( + DEFAULT_SYSTEM_PROMPT, + ApplicationConfigurationError, + run_task, +) from mini_code_agent.config import ( + AppSettings, ConfigurationError, default_config_path, load_settings, ) from mini_code_agent.diagnostics import DiagnosticReport, build_diagnostic_report from mini_code_agent.logging import configure_logging +from mini_code_agent.policy.models import SessionMode +from mini_code_agent.terminal import TerminalApprovalHandler, TerminalEventSink app = typer.Typer( name="mini-code-agent", @@ -64,6 +75,64 @@ def _render_report(report: DiagnosticReport) -> None: console.print(table) +def _load_command_settings(config: Path | None) -> AppSettings: + try: + settings = load_settings(config_path=config or default_config_path()) + except ConfigurationError as exc: + error_console.print(f"[red]Configuration error:[/red] {exc}") + raise typer.Exit(code=2) from exc + configured_secrets = ( + secret + for secret in (settings.anthropic_api_key, settings.openai_api_key) + if secret is not None + ) + configure_logging(settings.log_level.value, secrets=configured_secrets) + return settings + + +def _render_agent_result(result: AgentResult) -> None: + if result.final_text: + console.print(Text(result.final_text)) + console.print( + "[dim]" + f"Result: {result.stop_reason.value}; turns={result.turns}; " + f"tools={result.tool_calls}; " + f"tokens input={result.usage.input_tokens} output={result.usage.output_tokens}" + "[/dim]" + ) + if result.error: + error_console.print(f"[red]Agent stopped:[/red] {result.error}") + + +def _execute_task( + settings: AppSettings, + *, + workspace: Path, + user_prompt: str, + system_prompt: str, + session_mode: SessionMode, +) -> AgentResult: + approval = TerminalApprovalHandler( + console=console, + confirm=lambda prompt: typer.confirm(prompt, default=False), + ) + try: + return asyncio.run( + run_task( + settings, + workspace=workspace, + user_prompt=user_prompt, + system_prompt=system_prompt, + approval=approval, + session_mode=session_mode, + events=TerminalEventSink(console=console), + ) + ) + except ApplicationConfigurationError as exc: + error_console.print(f"[red]Configuration error:[/red] {exc}") + raise typer.Exit(code=2) from exc + + @app.command() def doctor( config: Annotated[ @@ -98,3 +167,85 @@ def doctor( _render_report(report) if not report.healthy: raise typer.Exit(code=1) + + +@app.command("run") +def run_agent_command( + task: Annotated[str, typer.Argument(help="Task for the coding agent.")], + workspace: Annotated[ + Path | None, + typer.Option("--workspace", "-w", help="Workspace directory. Defaults to the current one."), + ] = None, + config: Annotated[ + Path | None, + typer.Option("--config", help="Path to a TOML configuration file."), + ] = None, + system_prompt: Annotated[ + str | None, + typer.Option("--system-prompt", help="Override the built-in coding-agent instructions."), + ] = None, + non_interactive: Annotated[ + bool, + typer.Option( + "--non-interactive", + help="Deny writes and commands instead of prompting.", + ), + ] = False, +) -> None: + settings = _load_command_settings(config) + mode = SessionMode.NON_INTERACTIVE if non_interactive else SessionMode.INTERACTIVE + result = _execute_task( + settings, + workspace=workspace or Path.cwd(), + user_prompt=task, + system_prompt=system_prompt or DEFAULT_SYSTEM_PROMPT, + session_mode=mode, + ) + _render_agent_result(result) + if not result.succeeded: + raise typer.Exit(code=1) + + +@app.command() +def chat( + workspace: Annotated[ + Path | None, + typer.Option("--workspace", "-w", help="Workspace directory. Defaults to the current one."), + ] = None, + config: Annotated[ + Path | None, + typer.Option("--config", help="Path to a TOML configuration file."), + ] = None, + system_prompt: Annotated[ + str | None, + typer.Option("--system-prompt", help="Override the built-in coding-agent instructions."), + ] = None, +) -> None: + settings = _load_command_settings(config) + active_workspace = workspace or Path.cwd() + active_system_prompt = system_prompt or DEFAULT_SYSTEM_PROMPT + console.print( + "[dim]Interactive task mode. Each prompt starts an independent bounded run. " + "Use /exit or /quit to stop.[/dim]" + ) + failed = False + while True: + try: + task = typer.prompt("task") + except (EOFError, typer.Abort): + break + if task.strip().lower() in {"/exit", "/quit"}: + break + if not task.strip(): + continue + result = _execute_task( + settings, + workspace=active_workspace, + user_prompt=task, + system_prompt=active_system_prompt, + session_mode=SessionMode.INTERACTIVE, + ) + _render_agent_result(result) + failed = failed or not result.succeeded + if failed: + raise typer.Exit(code=1) diff --git a/src/mini_code_agent/config.py b/src/mini_code_agent/config.py index 5491c83..4c07206 100644 --- a/src/mini_code_agent/config.py +++ b/src/mini_code_agent/config.py @@ -22,6 +22,11 @@ class LogLevel(StrEnum): ERROR = "error" +class ProviderName(StrEnum): + OPENAI_COMPATIBLE = "openai_compatible" + ANTHROPIC = "anthropic" + + def default_data_dir() -> Path: return user_data_path("mini-code-agent", appauthor=False) @@ -40,6 +45,9 @@ class AppSettings(BaseModel): log_level: LogLevel = LogLevel.INFO data_dir: Path = Field(default_factory=default_data_dir) trace_enabled: bool = True + provider: ProviderName = ProviderName.OPENAI_COMPATIBLE + model: str | None = Field(default=None, min_length=1, max_length=256) + base_url: str | None = Field(default=None, min_length=1, max_length=2_048) anthropic_api_key: SecretStr | None = None openai_api_key: SecretStr | None = None @@ -48,6 +56,9 @@ def safe_dict(self) -> dict[str, object]: "log_level": self.log_level.value, "data_dir": str(self.data_dir), "trace_enabled": self.trace_enabled, + "provider": self.provider.value, + "model": self.model, + "base_url": self.base_url, "anthropic_api_key_configured": self.anthropic_api_key is not None, "openai_api_key_configured": self.openai_api_key is not None, } @@ -64,6 +75,9 @@ class EnvironmentSettings(BaseSettings): log_level: LogLevel | None = None data_dir: Path | None = None trace_enabled: bool | None = None + provider: ProviderName | None = None + model: str | None = Field(default=None, min_length=1, max_length=256) + base_url: str | None = Field(default=None, min_length=1, max_length=2_048) anthropic_api_key: SecretStr | None = None openai_api_key: SecretStr | None = None diff --git a/src/mini_code_agent/terminal.py b/src/mini_code_agent/terminal.py new file mode 100644 index 0000000..6d3329e --- /dev/null +++ b/src/mini_code_agent/terminal.py @@ -0,0 +1,77 @@ +from __future__ import annotations + +import os +import shlex +import subprocess +from collections.abc import Callable + +from rich.console import Console +from rich.syntax import Syntax +from rich.table import Table +from rich.text import Text + +from mini_code_agent.agent.events import ( + AgentEvent, + ModelStarted, + RunStarted, + RunStopped, + ToolStarted, +) +from mini_code_agent.policy.models import ApprovalRequest + + +class TerminalApprovalHandler: + def __init__( + self, + *, + console: Console, + confirm: Callable[[str], bool], + ) -> None: + self._console = console + self._confirm = confirm + + async def approve(self, request: ApprovalRequest) -> bool: + preview = request.preview + table = Table(title="Approval required", show_header=False) + table.add_column("Field", style="bold") + table.add_column("Value") + table.add_row("Tool", Text(preview.tool_name)) + table.add_row("Risk", Text(preview.risk.value)) + table.add_row("Action", Text(preview.summary)) + table.add_row("Reason", Text(preview.reason)) + if preview.resources: + table.add_row("Resources", Text("\n".join(preview.resources))) + if preview.command: + table.add_row("Command", Text(_format_argv(preview.command))) + table.add_row("Policy", Text(request.rationale)) + self._console.print(table) + if preview.diff: + self._console.print(Syntax(preview.diff, "diff", word_wrap=True)) + return self._confirm("Approve this action?") + + +class TerminalEventSink: + def __init__(self, *, console: Console) -> None: + self._console = console + + def publish(self, event: AgentEvent) -> None: + if isinstance(event, RunStarted): + self._console.print(f"[dim]Run started ({event.run_id})[/dim]") + elif isinstance(event, ModelStarted): + self._console.print(f"[dim]Model turn {event.turn}[/dim]") + elif isinstance(event, ToolStarted): + self._console.print(f"[dim]Tool: {event.tool_name}[/dim]") + elif isinstance(event, RunStopped): + self._console.print( + "[dim]" + f"Run {event.reason.value}; turns={event.turns}; " + f"tools={event.tool_calls}; " + f"tokens input={event.usage.input_tokens} output={event.usage.output_tokens}" + "[/dim]" + ) + + +def _format_argv(argv: tuple[str, ...]) -> str: + if os.name == "nt": + return subprocess.list2cmdline(argv) + return shlex.join(argv) diff --git a/tests/cli/test_cli.py b/tests/cli/test_cli.py index 8f726dd..d96df64 100644 --- a/tests/cli/test_cli.py +++ b/tests/cli/test_cli.py @@ -6,7 +6,11 @@ import pytest from typer.testing import CliRunner +from mini_code_agent.agent.models import AgentResult, StopReason +from mini_code_agent.application import ApplicationConfigurationError from mini_code_agent.cli import app +from mini_code_agent.policy.models import SessionMode +from mini_code_agent.providers.base import TokenUsage runner = CliRunner() @@ -19,15 +23,36 @@ def clear_project_environment(monkeypatch: pytest.MonkeyPatch) -> None: "MINI_CODE_AGENT_TRACE_ENABLED", "MINI_CODE_AGENT_ANTHROPIC_API_KEY", "MINI_CODE_AGENT_OPENAI_API_KEY", + "MINI_CODE_AGENT_PROVIDER", + "MINI_CODE_AGENT_MODEL", + "MINI_CODE_AGENT_BASE_URL", ): monkeypatch.delenv(name, raising=False) +def agent_result( + *, + reason: StopReason = StopReason.COMPLETED, + final_text: str | None = "Inspected the project.", + error: str | None = None, +) -> AgentResult: + return AgentResult( + run_id="run-1", + messages=(), + stop_reason=reason, + turns=1, + tool_calls=0, + usage=TokenUsage(input_tokens=10, output_tokens=4), + final_text=final_text, + error=error, + ) + + def test_version_option_prints_package_version() -> None: result = runner.invoke(app, ["--version"]) assert result.exit_code == 0 - assert result.stdout.strip() == "0.16.0a0" + assert result.stdout.strip() == "0.17.0a0" def test_module_entrypoint_prints_package_version() -> None: @@ -39,7 +64,7 @@ def test_module_entrypoint_prints_package_version() -> None: ) assert result.returncode == 0 - assert result.stdout.strip() == "0.16.0a0" + assert result.stdout.strip() == "0.17.0a0" def test_doctor_json_never_prints_secrets( @@ -110,3 +135,128 @@ def test_doctor_human_output_explains_unusable_data_path( assert "Data path usable" in result.stdout assert "Overall healthy" in result.stdout assert "False" in result.stdout + + +def test_run_executes_one_task_and_renders_result( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + calls: list[dict[str, object]] = [] + + async def fake_run_task(*args: object, **kwargs: object) -> AgentResult: + calls.append(dict(kwargs)) + return agent_result() + + monkeypatch.setattr("mini_code_agent.cli.run_task", fake_run_task) + monkeypatch.setenv("MINI_CODE_AGENT_MODEL", "Pro/zai-org/GLM-4.7") + monkeypatch.setenv("MINI_CODE_AGENT_OPENAI_API_KEY", "test-key") + + result = runner.invoke( + app, + ["run", "Inspect this project.", "--workspace", str(tmp_path)], + ) + + assert result.exit_code == 0 + assert "Inspected the project." in result.stdout + assert "completed" in result.stdout + assert calls[0]["workspace"] == tmp_path + assert calls[0]["user_prompt"] == "Inspect this project." + assert calls[0]["session_mode"] is SessionMode.INTERACTIVE + + +def test_run_non_interactive_forwards_policy_mode( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + modes: list[SessionMode] = [] + + async def fake_run_task(*args: object, **kwargs: object) -> AgentResult: + modes.append(kwargs["session_mode"]) # type: ignore[arg-type] + return agent_result() + + monkeypatch.setattr("mini_code_agent.cli.run_task", fake_run_task) + monkeypatch.setenv("MINI_CODE_AGENT_MODEL", "model-1") + monkeypatch.setenv("MINI_CODE_AGENT_OPENAI_API_KEY", "test-key") + + result = runner.invoke( + app, + [ + "run", + "Inspect.", + "--workspace", + str(tmp_path), + "--non-interactive", + ], + ) + + assert result.exit_code == 0 + assert modes == [SessionMode.NON_INTERACTIVE] + + +def test_run_configuration_error_exits_two_without_traceback( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + async def fake_run_task(*args: object, **kwargs: object) -> AgentResult: + raise ApplicationConfigurationError("Set MINI_CODE_AGENT_MODEL.") + + monkeypatch.setattr("mini_code_agent.cli.run_task", fake_run_task) + + result = runner.invoke( + app, + ["run", "Inspect.", "--workspace", str(tmp_path)], + ) + + assert result.exit_code == 2 + assert "Set MINI_CODE_AGENT_MODEL" in result.stderr + assert "Traceback" not in result.stderr + + +def test_run_agent_failure_exits_one_and_renders_public_error( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + async def fake_run_task(*args: object, **kwargs: object) -> AgentResult: + return agent_result( + reason=StopReason.PROVIDER_ERROR, + final_text=None, + error="Provider authentication failed.", + ) + + monkeypatch.setattr("mini_code_agent.cli.run_task", fake_run_task) + + result = runner.invoke( + app, + ["run", "Inspect.", "--workspace", str(tmp_path)], + ) + + assert result.exit_code == 1 + assert "Provider authentication failed." in result.stderr + assert "Traceback" not in result.stderr + + +def test_chat_runs_each_prompt_until_exit( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + prompts: list[str] = [] + + async def fake_run_task(*args: object, **kwargs: object) -> AgentResult: + prompts.append(str(kwargs["user_prompt"])) + return agent_result(final_text=f"Completed: {kwargs['user_prompt']}") + + monkeypatch.setattr("mini_code_agent.cli.run_task", fake_run_task) + monkeypatch.setenv("MINI_CODE_AGENT_MODEL", "model-1") + monkeypatch.setenv("MINI_CODE_AGENT_OPENAI_API_KEY", "test-key") + + result = runner.invoke( + app, + ["chat", "--workspace", str(tmp_path)], + input="Inspect files.\nSummarize changes.\n/exit\n", + ) + + assert result.exit_code == 0 + assert prompts == ["Inspect files.", "Summarize changes."] + assert "Completed: Inspect files." in result.stdout + assert "Completed: Summarize changes." in result.stdout + assert "independent bounded run" in result.stdout diff --git a/tests/integration/test_agent_loop.py b/tests/integration/test_agent_loop.py index 9b9404e..09b2c1e 100644 --- a/tests/integration/test_agent_loop.py +++ b/tests/integration/test_agent_loop.py @@ -53,7 +53,7 @@ async def test_fake_provider_drives_native_tool_call_round_trip() -> None: assert tool_result_message.role is MessageRole.USER assert tool_result_message.tool_results[0].tool_call_id == "call-1" payload = json.loads(tool_result_message.tool_results[0].content) - assert payload["package_version"] == "0.16.0a0" + assert payload["package_version"] == "0.17.0a0" assert [type(event) for event in events.events] == [ RunStarted, ModelStarted, diff --git a/tests/unit/test_application.py b/tests/unit/test_application.py new file mode 100644 index 0000000..a0c8df0 --- /dev/null +++ b/tests/unit/test_application.py @@ -0,0 +1,227 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import httpx +import pytest +from pydantic import SecretStr + +from mini_code_agent.agent.events import RecordingEventSink +from mini_code_agent.agent.models import StopReason +from mini_code_agent.application import ( + ApplicationConfigurationError, + build_provider, + build_tool_executor, + run_task, +) +from mini_code_agent.config import AppSettings, ProviderName +from mini_code_agent.domain.content import ToolCall +from mini_code_agent.domain.messages import Message +from mini_code_agent.policy.approval import StaticApprovalHandler +from mini_code_agent.policy.models import SessionMode +from mini_code_agent.providers.base import FinishReason, ModelRequest, ModelResponse +from mini_code_agent.providers.fake import ScriptedProvider + + +def settings_for(tmp_path: Path, **overrides: object) -> AppSettings: + return AppSettings.model_validate( + { + "data_dir": tmp_path / "data", + "provider": "openai_compatible", + "model": "Pro/zai-org/GLM-4.7", + "base_url": "https://api.siliconflow.cn/v1", + "openai_api_key": "test-key", + **overrides, + } + ) + + +def test_build_provider_requires_model_before_network_access(tmp_path: Path) -> None: + settings = settings_for(tmp_path, model=None) + + with pytest.raises(ApplicationConfigurationError, match="model"): + build_provider(settings) + + +@pytest.mark.parametrize( + ("provider", "settings_values", "message"), + [ + ( + ProviderName.OPENAI_COMPATIBLE, + {"openai_api_key": None}, + "MINI_CODE_AGENT_OPENAI_API_KEY", + ), + ( + ProviderName.ANTHROPIC, + { + "provider": "anthropic", + "openai_api_key": None, + "anthropic_api_key": None, + }, + "MINI_CODE_AGENT_ANTHROPIC_API_KEY", + ), + ], +) +def test_build_provider_requires_matching_api_key( + tmp_path: Path, + provider: ProviderName, + settings_values: dict[str, object], + message: str, +) -> None: + settings = settings_for(tmp_path, **settings_values) + assert settings.provider is provider + + with pytest.raises(ApplicationConfigurationError, match=message): + build_provider(settings) + + +@pytest.mark.asyncio +async def test_openai_compatible_provider_uses_siliconflow_endpoint(tmp_path: Path) -> None: + captured_urls: list[str] = [] + + async def handler(request: httpx.Request) -> httpx.Response: + captured_urls.append(str(request.url)) + assert request.headers["Authorization"] == "Bearer test-key" + return httpx.Response( + 200, + json={ + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 3, "completion_tokens": 1}, + }, + ) + + client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + provider = build_provider(settings_for(tmp_path), client=client) + try: + response = await provider.complete( + ModelRequest( + request_id="siliconflow-smoke", + system_prompt="", + messages=(Message.user_text("hello"),), + ) + ) + finally: + await provider.aclose() + await client.aclose() + + assert response.message.text == "ok" + assert captured_urls == ["https://api.siliconflow.cn/v1/chat/completions"] + + +@pytest.mark.asyncio +async def test_tool_executor_registers_cli_tools_and_asks_for_commands(tmp_path: Path) -> None: + approval = StaticApprovalHandler(approved=False) + executor = build_tool_executor( + tmp_path, + approval=approval, + session_mode=SessionMode.INTERACTIVE, + ) + + names = tuple(definition.name for definition in executor.definitions) + result = await executor.execute( + ToolCall( + id="command-1", + name="run_command", + arguments={ + "argv": ["python", "--version"], + "reason": "Check the Python version.", + }, + ) + ) + + assert names == ( + "read_file", + "search_text", + "write_file", + "edit_file", + "git_status", + "git_diff", + "run_command", + ) + assert json.loads(result.content)["error"]["code"] == "permission_denied" + assert len(approval.requests) == 1 + assert approval.requests[0].preview.command == ("python", "--version") + + +class ClosableScriptedProvider(ScriptedProvider): + def __init__(self, responses: list[ModelResponse]) -> None: + super().__init__(responses) + self.closed = False + + async def aclose(self) -> None: + self.closed = True + + +@pytest.mark.asyncio +async def test_run_task_executes_runtime_and_closes_provider(tmp_path: Path) -> None: + provider = ClosableScriptedProvider( + [ + ModelResponse( + message=Message.assistant_text("Inspected the project."), + finish_reason=FinishReason.STOP, + ) + ] + ) + events = RecordingEventSink() + + result = await run_task( + settings_for(tmp_path), + workspace=tmp_path, + user_prompt="Inspect this project.", + approval=StaticApprovalHandler(approved=False), + session_mode=SessionMode.INTERACTIVE, + events=events, + provider_factory=lambda settings: provider, + ) + + assert result.stop_reason is StopReason.COMPLETED + assert result.final_text == "Inspected the project." + assert provider.closed is True + assert [event.type for event in events.events] == [ + "run_started", + "model_started", + "model_completed", + "run_stopped", + ] + + +@pytest.mark.asyncio +async def test_run_task_closes_provider_when_runtime_fails(tmp_path: Path) -> None: + provider = ClosableScriptedProvider([]) + + result = await run_task( + settings_for(tmp_path), + workspace=tmp_path, + user_prompt="Inspect this project.", + approval=StaticApprovalHandler(approved=False), + session_mode=SessionMode.NON_INTERACTIVE, + provider_factory=lambda settings: provider, + ) + + assert result.stop_reason is StopReason.PROVIDER_ERROR + assert provider.closed is True + + +@pytest.mark.asyncio +async def test_build_provider_selects_anthropic(tmp_path: Path) -> None: + settings = settings_for( + tmp_path, + provider="anthropic", + model="claude-sonnet-4-5", + base_url=None, + openai_api_key=None, + anthropic_api_key=SecretStr("anthropic-key"), + ) + + provider = build_provider(settings) + try: + assert provider.__class__.__name__ == "AnthropicProvider" + finally: + await provider.aclose() diff --git a/tests/unit/test_config.py b/tests/unit/test_config.py index 15747eb..2bf5fc9 100644 --- a/tests/unit/test_config.py +++ b/tests/unit/test_config.py @@ -6,6 +6,7 @@ AppSettings, ConfigurationError, LogLevel, + ProviderName, load_settings, ) @@ -18,6 +19,9 @@ def clear_project_environment(monkeypatch: pytest.MonkeyPatch) -> None: "MINI_CODE_AGENT_TRACE_ENABLED", "MINI_CODE_AGENT_ANTHROPIC_API_KEY", "MINI_CODE_AGENT_OPENAI_API_KEY", + "MINI_CODE_AGENT_PROVIDER", + "MINI_CODE_AGENT_MODEL", + "MINI_CODE_AGENT_BASE_URL", ): monkeypatch.delenv(name, raising=False) @@ -27,6 +31,9 @@ def test_defaults_are_valid_without_a_config_file(tmp_path: Path) -> None: assert settings.log_level is LogLevel.INFO assert settings.trace_enabled is True + assert settings.provider is ProviderName.OPENAI_COMPATIBLE + assert settings.model is None + assert settings.base_url is None assert settings.anthropic_api_key is None assert settings.openai_api_key is None @@ -62,6 +69,9 @@ def test_safe_dict_never_contains_secret_values(tmp_path: Path) -> None: settings = AppSettings.model_validate( { "data_dir": tmp_path, + "provider": "openai_compatible", + "model": "Pro/zai-org/GLM-4.7", + "base_url": "https://api.siliconflow.cn/v1", "anthropic_api_key": "anthropic-secret", "openai_api_key": "openai-secret", } @@ -72,10 +82,38 @@ def test_safe_dict_never_contains_secret_values(tmp_path: Path) -> None: assert "anthropic-secret" not in rendered assert "openai-secret" not in rendered + assert payload["provider"] == "openai_compatible" + assert payload["model"] == "Pro/zai-org/GLM-4.7" + assert payload["base_url"] == "https://api.siliconflow.cn/v1" assert payload["anthropic_api_key_configured"] is True assert payload["openai_api_key_configured"] is True +def test_provider_environment_overrides_file_values( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + config_path = tmp_path / "config.toml" + config_path.write_text( + """ +[mini_code_agent] +provider = "anthropic" +model = "claude-from-file" +base_url = "https://provider-from-file.example" +""".strip(), + encoding="utf-8", + ) + monkeypatch.setenv("MINI_CODE_AGENT_PROVIDER", "openai_compatible") + monkeypatch.setenv("MINI_CODE_AGENT_MODEL", "Pro/zai-org/GLM-4.7") + monkeypatch.setenv("MINI_CODE_AGENT_BASE_URL", "https://api.siliconflow.cn/v1") + + settings = load_settings(config_path=config_path) + + assert settings.provider is ProviderName.OPENAI_COMPATIBLE + assert settings.model == "Pro/zai-org/GLM-4.7" + assert settings.base_url == "https://api.siliconflow.cn/v1" + + def test_invalid_toml_section_has_an_actionable_error(tmp_path: Path) -> None: config_path = tmp_path / "config.toml" config_path.write_text("mini_code_agent = 42", encoding="utf-8") diff --git a/tests/unit/test_package.py b/tests/unit/test_package.py index dae4dfc..f39e737 100644 --- a/tests/unit/test_package.py +++ b/tests/unit/test_package.py @@ -4,7 +4,7 @@ def test_package_exports_release_version() -> None: - assert __version__ == "0.16.0a0" + assert __version__ == "0.17.0a0" def test_package_includes_pep561_marker() -> None: diff --git a/tests/unit/test_terminal.py b/tests/unit/test_terminal.py new file mode 100644 index 0000000..f32671f --- /dev/null +++ b/tests/unit/test_terminal.py @@ -0,0 +1,121 @@ +from __future__ import annotations + +import os +import shlex +import subprocess + +import pytest +from rich.console import Console + +from mini_code_agent.agent.events import ModelStarted, RunStarted, RunStopped, ToolStarted +from mini_code_agent.agent.models import StopReason +from mini_code_agent.policy.models import ( + ActionPreview, + ApprovalRequest, + RiskLevel, +) +from mini_code_agent.providers.base import TokenUsage +from mini_code_agent.terminal import TerminalApprovalHandler, TerminalEventSink +from mini_code_agent.tools.base import SideEffect + + +def recording_console() -> Console: + return Console(record=True, width=100, color_system=None) + + +@pytest.mark.asyncio +async def test_approval_handler_renders_action_details_and_returns_decision() -> None: + console = recording_console() + prompts: list[str] = [] + + def confirm(prompt: str) -> bool: + prompts.append(prompt) + return True + + handler = TerminalApprovalHandler(console=console, confirm=confirm) + request = ApprovalRequest( + preview=ActionPreview( + tool_call_id="edit-1", + tool_name="edit_file", + side_effect=SideEffect.WRITE, + risk=RiskLevel.MEDIUM, + summary="Edit one workspace file.", + reason="Implement [red]the requested[/red] behavior.", + resources=("src/app.py",), + command=("python", "script with spaces.py", "--name=a b"), + diff="--- a/src/app.py\n+++ b/src/app.py\n-old\n+new\n", + ), + rule_id="default-write", + rationale="Write tools require approval by default.", + ) + + approved = await handler.approve(request) + output = console.export_text() + + assert approved is True + assert prompts == ["Approve this action?"] + assert "edit_file" in output + assert "medium" in output + assert "Implement [red]the requested[/red] behavior." in output + assert "src/app.py" in output + command = request.preview.command + assert command is not None + expected_command = subprocess.list2cmdline(command) if os.name == "nt" else shlex.join(command) + assert expected_command in output + assert "-old" in output + assert "+new" in output + + +@pytest.mark.asyncio +async def test_approval_handler_can_deny_action() -> None: + handler = TerminalApprovalHandler( + console=recording_console(), + confirm=lambda prompt: False, + ) + request = ApprovalRequest( + preview=ActionPreview( + tool_call_id="write-1", + tool_name="write_file", + side_effect=SideEffect.WRITE, + risk=RiskLevel.MEDIUM, + summary="Create one workspace file.", + ), + rule_id="default-write", + rationale="Write tools require approval by default.", + ) + + assert await handler.approve(request) is False + + +def test_event_sink_renders_bounded_lifecycle_metadata_only() -> None: + console = recording_console() + sink = TerminalEventSink(console=console) + + sink.publish(RunStarted(run_id="run-1", max_turns=8)) + sink.publish(ModelStarted(run_id="run-1", turn=1, request_id="request-1")) + sink.publish( + ToolStarted( + run_id="run-1", + turn=1, + tool_call_id="call-1", + tool_name="read_file", + side_effect=SideEffect.READ_ONLY, + ) + ) + sink.publish( + RunStopped( + run_id="run-1", + turns=1, + reason=StopReason.COMPLETED, + tool_calls=1, + usage=TokenUsage(input_tokens=10, output_tokens=4), + ) + ) + output = console.export_text() + + assert "Run started" in output + assert "Model turn 1" in output + assert "read_file" in output + assert "completed" in output + assert "input=10" in output + assert "secret prompt" not in output diff --git a/tests/unit/tools/test_runtime_info.py b/tests/unit/tools/test_runtime_info.py index 1fede63..e3cbd53 100644 --- a/tests/unit/tools/test_runtime_info.py +++ b/tests/unit/tools/test_runtime_info.py @@ -35,7 +35,7 @@ async def test_runtime_info_returns_safe_structured_data() -> None: payload = json.loads(result.content) assert result.tool_call_id == "call-1" assert result.is_error is False - assert payload["package_version"] == "0.16.0a0" + assert payload["package_version"] == "0.17.0a0" assert payload["python_version"] assert payload["platform"] diff --git a/uv.lock b/uv.lock index 3665e8f..5af5b56 100644 --- a/uv.lock +++ b/uv.lock @@ -360,7 +360,7 @@ wheels = [ [[package]] name = "mini-code-agent" -version = "0.16.0a0" +version = "0.17.0a0" source = { editable = "." } dependencies = [ { name = "defusedxml" },