diff --git a/CHANGELOG.md b/CHANGELOG.md index 1385f34..b8ea0dc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,12 @@ All notable changes follow Keep a Changelog. Versions follow Semantic Versioning ### Added +- Loopback-only `mini-code-agent web` command with a responsive three-pane local workbench, + real-time SSE lifecycle activity, task cancellation, and browser approval for governed actions. +- Bounded in-memory Web run manager with one active run, monotonic replayable events, + Future-based single-use approvals, and deterministic cancellation cleanup. +- Learning and resume documentation for the Web adapter, asyncio/SSE flow, browser trust boundary, + and Java backend concept mapping. - Provider-backed `run` and `chat` CLI commands that compose the existing Agent runtime, OpenAI-compatible or Anthropic adapters, bounded Workspace, and governed built-in tools. - SiliconFlow configuration through `provider`, `model`, `base_url`, and the existing @@ -15,6 +21,10 @@ All notable changes follow Keep a Changelog. Versions follow Semantic Versioning ### Security +- Web requests that mutate state require a process-random token; browser Origins must be + loopback, CORS is not enabled, and the CLI rejects remote binding. +- Workspace selection and Provider credentials stay server-side. Browser-rendered model text, + paths, commands, and diffs use text nodes rather than dynamic HTML. - Read-only tools remain allowed by default. Writes and CLI-enabled command execution require an interactive approval; non-interactive mode denies both without prompting. - CLI output uses normalized public errors and never renders API key values. Live provider calls @@ -24,6 +34,9 @@ All notable changes follow Keep a Changelog. Versions follow Semantic Versioning ### Verification +- M8 local Windows verification passed 1218 tests with 13 privilege/platform skips and 88.52% + branch-aware package coverage. Ruff format/check, strict Pyright, and browser layout checks at + 1440x1024, 1024x768, and 390x844 passed. - Local uv-managed Python 3.13.14 passed 1201 tests with 13 Windows privilege/platform skips and 88.56% branch-aware package coverage. Ruff format/check and strict Pyright passed. - MockTransport verified the SiliconFlow-compatible diff --git a/README.md b/README.md index 354fbe2..aeb48db 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ A framework-light, provider-neutral coding agent built from first principles. -> Status: pre-alpha. M7 provides a provider-neutral Agent Core, Anthropic/OpenAI-compatible +> Status: pre-alpha. M8 provides a provider-neutral Agent Core, Anthropic/OpenAI-compatible > adapters, a schema-validating Tool Registry, a cross-platform Workspace boundary, bounded > Read/Search, conflict-aware Write/Edit, policy-governed argv command execution, and deterministic > context admission, hardened read-only Git evidence, governed Pytest diagnostics, versioned SQLite @@ -10,7 +10,8 @@ A framework-light, provider-neutral coding agent built from first principles. > loop, provenance-aware lazy Skills, deterministic host-registered Tool Hooks, and host-pinned > local MCP stdio Tools, bounded host-profiled read-only analysis Subagents, and host-managed > Worktree implementation candidates with separately approved adoption, plus provider-backed -> `run` and `chat` terminal commands with governed action previews. OS sandboxing, +> `run` and `chat` terminal commands plus a loopback-only Web console with live activity, +> governed action previews, approval, and cancellation. OS sandboxing, > shell-string execution, project-provided executable Hooks, automatic Repair resume, remote > HTTP/OAuth MCP, automatic commit/merge/push, and live-provider CI are not implemented. @@ -74,6 +75,21 @@ Start an interactive task loop: mini-code-agent chat --config .\config.toml --workspace . ``` +Start the local Web console: + +```powershell +mini-code-agent web --config .\config.toml --workspace . +``` + +The browser opens `http://127.0.0.1:8765` by default. Use `--no-open` to start only the server or +`--port` to choose another local port. The command rejects non-loopback hosts. The Workspace is +fixed when the process starts; it cannot be changed from the browser. + +The Web console shows Agent lifecycle events, Tool activity, token usage, bounded action previews, +and file diffs. Write and command actions pause until they are approved or rejected in the +inspector. API keys stay in server-side settings; the browser receives only a configured/not +configured flag and a process-random request token. + Each `chat` prompt starts an independent bounded Agent run against the same workspace; durable conversation memory is not implied. Read-only tools run automatically. File writes and local argv commands display an action preview and require explicit confirmation. Use `--non-interactive` with diff --git a/config.example.toml b/config.example.toml index 56cc62d..6fa07ef 100644 --- a/config.example.toml +++ b/config.example.toml @@ -8,3 +8,6 @@ base_url = "https://api.siliconflow.cn/v1" # Keep API keys out of this file. For SiliconFlow, set: # MINI_CODE_AGENT_OPENAI_API_KEY +# +# Then start the local UI with: +# mini-code-agent web --config .\config.example.toml --workspace . diff --git a/docs/learning/m8-web-console.md b/docs/learning/m8-web-console.md new file mode 100644 index 0000000..868caa3 --- /dev/null +++ b/docs/learning/m8-web-console.md @@ -0,0 +1,115 @@ +# M8 学习笔记:把 Agent Runtime 变成可交互的本地 Web 产品 + +## 1. 本阶段解决的问题 + +M7 已经能通过命令行调用真实模型,但运行过程、工具活动、审批和 diff 都依赖终端展示。 +M8 增加一个本地 Web 适配层,不重写 Agent Runtime: + +```text +浏览器 + -> FastAPI REST(启动、审批、取消) + -> WebRunManager + -> run_task() + -> AgentRuntime -> Provider / Tool / Policy + -> EventSink / ApprovalHandler + -> SSE -> 浏览器活动面板 +``` + +产品目标是让用户看见 Agent 正在做什么,并在有副作用的动作执行前做决定。Web 层是 +交互适配器,不是新的 Agent 框架。 + +## 2. 前置知识与本项目知识点 + +| 学习主题 | 先掌握什么 | 在项目中的落点 | +|---|---|---| +| Python 异步 | coroutine、Task、Future、取消传播 | 后台 Agent run、待审批 Future、取消与清理 | +| FastAPI | 路由、依赖、Middleware、StreamingResponse | REST、CSRF 校验、Origin 校验、SSE | +| Pydantic | frozen model、字段上下界、序列化 | Web 请求、运行快照、事件信封 | +| 浏览器基础 | fetch、EventSource、DOM API、响应式 CSS | 启动任务、消费事件、安全渲染、移动端抽屉 | +| Agent Harness | EventSink、ApprovalHandler、Policy | 把已有运行时能力映射到 Web,而不绕过治理 | +| Web 安全 | secret boundary、CSRF、Origin、CSP、XSS | Key 留在服务端、随机令牌、文本节点渲染 | + +建议边做边补,不需要先系统学完前端或 FastAPI。 + +## 3. 核心实现 + +### 3.1 WebRunManager + +`WebRunManager` 是 Web 层的运行状态机: + +- 一次只允许一个活跃任务,避免多个浏览器操作争用同一工作区; +- 使用递增 `sequence` 给事件排序,并在有界 `deque` 中保留最近事件; +- 浏览器断线后通过 `after` 序号重放事件; +- 完成、失败和取消都产生明确终态; +- 生命周期事件不包含用户 prompt、Tool 参数/结果或 API Key。 + +这与 Flink JobManager 的相似点是都管理任务生命周期和状态;差异是这里是单进程、 +单活跃任务,没有分布式容错和 Checkpoint 语义。 + +### 3.2 Future 驱动的人工审批 + +当 `GovernedToolExecutor` 请求审批时,`WebApprovalHandler`: + +1. 创建 `asyncio.Future[bool]`; +2. 发布有界 `approval_required` 事件; +3. Agent 协程等待 Future,不执行工具; +4. 浏览器通过 REST 提交允许或拒绝; +5. Manager 先移除 Future,再设置结果,保证决定只能使用一次。 + +这类似 Java 的 `CompletableFuture`:生产者暂停等待外部决策,另一个请求处理器 +完成 Future。取消任务时所有待审批 Future 都会被拒绝并清理。 + +### 3.3 SSE 为什么适合这里 + +浏览器到服务端只有启动、审批和取消三个低频命令,服务端到浏览器则持续发送运行事件。 +SSE 提供单向事件流、浏览器原生 `EventSource` 和自动重连,不需要为双向 WebSocket +协议增加额外状态。事件带序号,重连时可以从最后位置继续。 + +当前不是逐 Token 流式输出。Provider 完成后才展示最终文本,SSE 传输的是 Agent +生命周期和工具活动。 + +### 3.4 浏览器和服务端的信任边界 + +- CLI 启动时固定 Workspace,浏览器不能传入任意路径; +- CLI 只允许回环 Host,不能使用 `0.0.0.0` 暴露到局域网; +- API Key 只由服务端配置读取,bootstrap 只返回布尔状态; +- 修改请求必须携带进程随机令牌,并通过 loopback Origin 检查; +- 模型文本、命令、路径和 diff 通过 `textContent` 显示; +- CSP 禁止第三方脚本和页面嵌入。 + +这些措施降低本地浏览器攻击面,但 Workspace/Policy 仍不是 OS Sandbox。 + +## 4. Java 后端经验映射 + +| Java / 数据开发概念 | Python / M8 对应 | +|---|---| +| Spring MVC Controller | FastAPI 路由函数 | +| HandlerInterceptor / Filter | FastAPI Middleware 和依赖 | +| CompletableFuture | `asyncio.Future` | +| ExecutorService Future.cancel | `asyncio.Task.cancel()` 和取消传播 | +| WebFlux ServerSentEvent | `StreamingResponse` + `text/event-stream` | +| DTO + Bean Validation | frozen Pydantic model + Field 约束 | +| ConcurrentHashMap 中的运行状态 | event-loop 内的 run/pending 字典 | +| Flink event-time sequence / offset | WebEvent sequence 与断线重放 | +| SQL 权限审批 | Tool Policy + 一次性 ActionPreview 审批 | + +Python 的关键差异是:同一事件循环中的共享状态通常不需要线程锁,但不能在协程中执行 +阻塞 I/O;取消是协作式异常传播,必须在 `finally` 中清理资源。 + +## 5. 建议学习练习 + +1. 从 `POST /api/runs` 跟到 `AgentRuntime.run()`,画出对象创建和调用顺序。 +2. 跟踪一次 `run_command`:Policy ASK -> Future -> 浏览器允许 -> Tool 执行。 +3. 删除 CSRF Header 或改成外部 Origin,观察接口为什么返回 403。 +4. 运行中刷新页面,解释 bootstrap 的 `active_run` 与 SSE `after` 如何恢复界面。 +5. 为 Manager 增加事件保留边界测试,说明慢客户端不会让内存无限增长。 +6. 对比 SSE 和 WebSocket,说明本项目为什么暂时不需要双向长连接。 + +## 6. 当前边界 + +- 仅支持本地单用户、单工作区、单活跃任务; +- 运行和会话只保存在内存中,进程退出后不恢复 Web 状态; +- 最终回答不是逐 Token 流式输出; +- 当前未把 Skills、MCP、Subagent、Worktree candidate 组合进 Web composition root; +- 图像生成 API 尚未接入,后续应作为受治理 Tool,而不是让浏览器直接持有 Key; +- 自动化测试使用 Mock/Scripted Provider,不声明已完成真实 SiliconFlow smoke。 diff --git a/docs/learning/progress.md b/docs/learning/progress.md index 49062be..c500bea 100644 --- a/docs/learning/progress.md +++ b/docs/learning/progress.md @@ -15,6 +15,7 @@ | L10 MCP | Complete and released | Governed stdio, exact grants, real SDK integration; v0.14 evidence | | L11 Subagent and Worktree | Complete and released | Host-profiled analysis plus governed Worktree candidates/adoption; v0.16 evidence | | L12 CI, benchmark and release | In progress | v0.16 prerelease and cross-platform evidence complete; benchmark remains separate | +| L13 Local Web console | Complete locally | FastAPI adapter, SSE event replay, Future approval, responsive browser workbench | ## L0 Notes diff --git a/docs/resume/m8-web-console-profile.md b/docs/resume/m8-web-console-profile.md new file mode 100644 index 0000000..df4594e --- /dev/null +++ b/docs/resume/m8-web-console-profile.md @@ -0,0 +1,109 @@ +# Mini CodeAgent M8 简历与面试说明 + +## 项目介绍 + +Mini CodeAgent 是一个使用 Python 从零实现的 Coding Agent Harness。项目实现了 +Model -> ToolCall -> ToolResult -> Model 的有界循环,并把 Workspace、文件/Git/命令工具、 +Policy、人工审批、上下文预算和 Provider Adapter 拆成可测试模块。M8 在不改动核心运行时 +的前提下增加本地 Web 工作台,用于提交项目任务、观察运行活动、审批副作用操作和取消任务。 + +## 技术栈 + +- Python 3.12/3.13、asyncio、强类型 Protocol +- FastAPI、Uvicorn、Pydantic +- Server-Sent Events、REST、HTML/CSS、原生 JavaScript +- OpenAI-compatible Chat Completions、SiliconFlow 配置 +- Typer、httpx、Pytest、pytest-asyncio +- Ruff、Pyright、GitHub Actions + +## 30 秒面试介绍 + +“我实现了一个 Python Mini Coding Agent,不只是调用一次大模型,而是完整实现有界 +Agent Loop、Tool Calling、工作区边界和有副作用工具审批。为了让运行过程可观察,我又做了 +一个只绑定本机回环地址的 Web 工作台。FastAPI 负责启动、审批和取消接口,SSE 推送模型与 +工具生命周期事件;命令或写文件时,Agent 通过 asyncio Future 暂停,用户查看资源、argv +和 diff 后只能批准一次。API Key 和 Workspace 都留在服务端,前端只接收必要状态并用文本 +节点渲染模型内容。测试用 Scripted Provider 和 ASGITransport,不依赖真实额度。” + +## 项目亮点 + +### 1. 核心运行时与 Web 交互解耦 + +**为什么使用:** CLI 和 Web 的输入输出方式不同,但 Provider、Tool、Policy 和 Agent Loop +不应该复制两套。 + +**技术实现:** `run_task()` 作为 Composition Root;Web 层只实现 `EventSink` 和 +`ApprovalHandler` 两个协议,再通过依赖注入调用已有 Runtime。 + +**实现功能:** 同一套 Agent 能从终端或浏览器运行,并共享相同工具治理语义。 + +**解决问题:** 避免界面层绕过 Policy,也降低新增交互入口时的重复代码和行为漂移。 + +**代码证据:** `src/mini_code_agent/application.py`、 +`src/mini_code_agent/web/manager.py`、`src/mini_code_agent/web/app.py`。 + +### 2. SSE 可观察运行与有界事件重放 + +**为什么使用:** 运行事件主要是服务端单向推送,WebSocket 的双向协议复杂度在这里没有收益。 + +**技术实现:** Manager 为事件分配单调递增 sequence,使用有界 deque 保留事件; +FastAPI `StreamingResponse` 输出具名 SSE,浏览器用 `EventSource` 消费并按序号重连。 + +**实现功能:** 展示模型调用、Tool 开始/结束、Token 用量、完成/失败/取消状态。 + +**解决问题:** 终端黑盒运行变成可追踪时间线;短暂断线后可恢复最近事件,同时限制内存增长。 + +**代码证据:** `WebRunManager.subscribe()`、`run_events()`、`static/app.js`。 + +### 3. Future 驱动的一次性人工审批 + +**为什么使用:** 模型提出命令或写入动作不等于用户授权,Web 请求和 Agent 协程又是两个独立 +控制流。 + +**技术实现:** 每个 ToolCall 创建一个 `asyncio.Future[bool]`;审批事件包含有界 +ActionPreview;REST 决策先从 pending map 移除 Future 再完成它,重复和过期决定返回冲突。 + +**实现功能:** Agent 在副作用执行前暂停,用户查看工具、风险、原因、资源、argv 和 diff 后 +允许一次或拒绝;取消任务会拒绝并清理全部待审批。 + +**解决问题:** 防止模型自授权、重复点击和过期审批触发工具,保证取消不会遗留悬挂协程。 + +**代码证据:** `_WebApprovalHandler`、`decide_approval()`、`cancel()` 及 Manager 单元测试。 + +### 4. 本地 Web 的服务端信任边界 + +**为什么使用:** Coding Agent 拥有本地文件和进程能力,普通“localhost 页面”仍需防止 +远程绑定、跨站请求和模型内容注入。 + +**技术实现:** CLI 拒绝非 loopback Host;Workspace 启动时固定;修改接口校验随机请求令牌 +和 loopback Origin;Key 使用服务端 `SecretStr`/环境变量;设置 CSP;动态内容只写 +`textContent`。 + +**实现功能:** 浏览器可以操作 Agent,但不能选择任意服务器路径、读取 Key 或从外部站点静默 +发起批准请求。 + +**解决问题:** 缩小本地管理界面的攻击面,并明确 Web 治理与 OS Sandbox 的边界。 + +**代码证据:** `create_web_app()` Middleware/依赖、`web()` Host 校验、静态资源契约测试。 + +### 5. 确定性的无凭证测试 + +**为什么使用:** 真实模型存在费用、网络波动和非确定性,不适合作为默认 CI 前提。 + +**技术实现:** Manager 注入 async runner;API 使用 `httpx.ASGITransport`;核心 Agent 使用 +Scripted Provider/MockTransport;前端用静态契约和浏览器响应式检查。 + +**实现功能:** 覆盖运行冲突、事件顺序、密钥脱敏、审批、取消、CSRF、Origin、SSE 和 CLI。 + +**解决问题:** 在不消耗 Token、不上传凭证的情况下验证控制流和协议边界。 + +**代码证据:** `tests/unit/web/`、`tests/cli/test_cli.py`。 + +## 诚实边界 + +- 不描述为 Claude Code 的完整替代品; +- 不声称是多用户或可公网部署的 Agent 平台; +- 不把 loopback、Workspace 或 Policy 描述为 OS Sandbox; +- 不声称默认 CI 验证了真实 SiliconFlow 账户; +- 不编造效率、准确率或成本下降百分比; +- 图像生成尚未接入当前 Agent 工具链。 diff --git a/docs/superpowers/plans/2026-07-02-m8-web-console.md b/docs/superpowers/plans/2026-07-02-m8-web-console.md new file mode 100644 index 0000000..361780f --- /dev/null +++ b/docs/superpowers/plans/2026-07-02-m8-web-console.md @@ -0,0 +1,250 @@ +# M8 Local Web Console Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Build a secure loopback-only Web console for running, observing, approving, and cancelling Mini CodeAgent tasks. + +**Architecture:** Add a FastAPI adapter around the existing `run_task`, bridge Agent events and approvals through a bounded in-memory run manager, and serve a dependency-free three-pane frontend from package resources. Keep Provider secrets and Workspace selection exclusively server-side. + +**Tech Stack:** Python 3.12/3.13, FastAPI, Uvicorn, asyncio, Pydantic, SSE, HTML, CSS, browser-native JavaScript, pytest, httpx ASGITransport + +--- + +### Task 1: Web Contracts and Run Manager + +**Files:** +- Create: `src/mini_code_agent/web/__init__.py` +- Create: `src/mini_code_agent/web/models.py` +- Create: `src/mini_code_agent/web/manager.py` +- Create: `tests/unit/web/__init__.py` +- Create: `tests/unit/web/test_models.py` +- Create: `tests/unit/web/test_manager.py` + +- [ ] **Step 1: Write failing model and lifecycle tests** + +Specify bounded `StartRunRequest`, `ApprovalDecisionRequest`, `WebEvent`, `RunSnapshot`, and +`WebRunManager`. Tests assert one active run, monotonic sequence values, terminal result events, +and no prompt or API-key values in lifecycle payloads. + +- [ ] **Step 2: Verify RED** + +Run: + +```powershell +python -m pytest tests/unit/web/test_models.py tests/unit/web/test_manager.py -q +``` + +Expected: collection fails because `mini_code_agent.web` does not exist. + +- [ ] **Step 3: Implement models and basic run lifecycle** + +Use frozen Pydantic models with bounded strings. The manager accepts an injected async runner: + +```python +type TaskRunner = Callable[ + [str, ApprovalHandler, EventSink], + Awaitable[AgentResult], +] +``` + +Publish `web_run_started`, normalized `agent_event`, and `web_run_completed` envelopes into a +bounded retained deque and subscriber queues. + +- [ ] **Step 4: Verify GREEN** + +Run the Task 1 tests and expect all to pass. + +### Task 2: Browser Approval and Cancellation + +**Files:** +- Modify: `src/mini_code_agent/web/manager.py` +- Modify: `tests/unit/web/test_manager.py` + +- [ ] **Step 1: Write failing approval tests** + +Start a runner that calls `approval.approve()`. Assert the manager publishes a bounded +`approval_required` event, ignores unknown decisions, resolves approve/reject once, and rejects a +second decision. + +- [ ] **Step 2: Verify RED** + +Run the approval tests and expect missing `decide_approval`. + +- [ ] **Step 3: Implement pending approval ownership** + +Store one Future per `(run_id, tool_call_id)`. Remove it before resolving. Cancellation resolves +pending approvals as rejected, cancels the task, awaits task completion, and emits one terminal +event. + +- [ ] **Step 4: Verify GREEN** + +Run manager tests and expect all to pass. + +### Task 3: FastAPI Application and Security + +**Files:** +- Modify: `pyproject.toml` +- Modify: `uv.lock` +- Create: `src/mini_code_agent/web/app.py` +- Create: `tests/unit/web/test_app.py` + +- [ ] **Step 1: Add FastAPI runtime dependency** + +Add: + +```toml +"fastapi>=0.116,<1", +"uvicorn>=0.34,<1", +``` + +Run `uv lock` and `uv sync --locked --all-groups`. + +- [ ] **Step 2: Write failing ASGI tests** + +Using `httpx.ASGITransport`, assert bootstrap redacts keys, health works, mutating requests require +the token, non-loopback Origin is rejected, one run starts, a second conflicts, SSE replays, and +approval/cancel routes delegate to the manager. + +- [ ] **Step 3: Verify RED** + +Run `python -m pytest tests/unit/web/test_app.py -q`. + +Expected: import failure because `web.app` does not exist. + +- [ ] **Step 4: Implement the application factory** + +Create: + +```python +def create_web_app( + settings: AppSettings, + *, + workspace: Path, + manager: WebRunManager | None = None, + csrf_token: str | None = None, +) -> FastAPI: + ... +``` + +Use fixed static resource routes and `StreamingResponse` with `text/event-stream`. + +- [ ] **Step 5: Verify GREEN** + +Run all Web unit tests. + +### Task 4: Three-Pane Frontend + +**Files:** +- Create: `src/mini_code_agent/web/static/index.html` +- Create: `src/mini_code_agent/web/static/styles.css` +- Create: `src/mini_code_agent/web/static/app.js` +- Create: `tests/unit/web/test_static.py` + +- [ ] **Step 1: Write failing static contract tests** + +Assert the package contains the three files, no external CDN references, no inline API key field, +required landmarks and controls exist, and JavaScript uses `textContent` rather than dynamic +`innerHTML`. + +- [ ] **Step 2: Verify RED** + +Run static tests and expect missing resources. + +- [ ] **Step 3: Implement semantic HTML and stable layout** + +Create the top bar, session rail, transcript, composer, inspector tabs, activity timeline, +approval panel, diff viewer, empty/error/running states, and mobile inspector drawer. + +- [ ] **Step 4: Implement browser behavior** + +Fetch bootstrap, start tasks, consume SSE, render events, submit approval decisions, cancel the +active run, reconnect by sequence, and update stable control states. + +- [ ] **Step 5: Verify GREEN** + +Run static and ASGI tests. + +### Task 5: Web CLI + +**Files:** +- Modify: `src/mini_code_agent/cli.py` +- Modify: `tests/cli/test_cli.py` + +- [ ] **Step 1: Write failing CLI tests** + +Patch `uvicorn.run` and `webbrowser.open`. Assert `web` forwards the app, host, port, and log level; +rejects non-loopback hosts; supports `--no-open`; and maps configuration errors to exit code 2. + +- [ ] **Step 2: Verify RED** + +Run CLI tests and expect no `web` command. + +- [ ] **Step 3: Implement `mini-code-agent web`** + +Load settings, create the Web app, optionally open the loopback URL, and run Uvicorn. Permit only +`127.0.0.1`, `localhost`, and `::1`. + +- [ ] **Step 4: Verify GREEN** + +Run CLI and Web tests. + +### Task 6: Documentation and Release Metadata + +**Files:** +- Modify: `README.md` +- Modify: `CHANGELOG.md` +- Modify: `config.example.toml` +- Create: `docs/learning/m8-web-console.md` +- Create: `docs/resume/m8-web-console-profile.md` +- Modify: `pyproject.toml` +- Modify: `uv.lock` +- Modify: version assertions under `tests/` + +- [ ] **Step 1: Document local launch** + +Document environment-only API key setup, `mini-code-agent web --workspace .`, loopback boundary, +approval flow, cancellation, and current limitations. + +- [ ] **Step 2: Add learning and interview material** + +Explain SSE, Future-based approval, CSRF, loopback binding, server/browser trust boundaries, and +Java/Spring mappings. Add resume-safe highlights with code/test evidence and no invented metrics. + +- [ ] **Step 3: Bump version** + +Set package version and assertions to `0.18.0a0`. + +- [ ] **Step 4: Verify focused suite** + +Run all Web, CLI, package, and runtime-info tests. + +### Task 7: Browser and Full Verification + +**Files:** +- Verify only + +- [ ] **Step 1: Run quality gates** + +```powershell +python -m ruff format --check . +python -m ruff check . +pyright --pythonpath .\.venv\Scripts\python.exe +python -m pytest --cov -q +``` + +Expected: zero format/lint/type errors, all runnable tests pass, coverage remains above 85%. + +- [ ] **Step 2: Start the local server** + +Launch `mini-code-agent web --workspace . --config config.example.toml --no-open` in a hidden +background process on a free loopback port. + +- [ ] **Step 3: Browser QA** + +Use the in-app browser at 1440x1024, 1024x768, and 390x844. Verify nonblank pixels, fixed layout, +no overlap, responsive collapse, tabs, composer, simulated run events, approval state, and diff. + +- [ ] **Step 4: Final safety review** + +Confirm no API key in Git diff, no remote bind, no CORS, no dynamic HTML injection, no browser +Workspace input, bounded queues, single-use approvals, and clean shutdown. diff --git a/docs/superpowers/specs/2026-07-02-m8-web-console-design.md b/docs/superpowers/specs/2026-07-02-m8-web-console-design.md new file mode 100644 index 0000000..c678df8 --- /dev/null +++ b/docs/superpowers/specs/2026-07-02-m8-web-console-design.md @@ -0,0 +1,136 @@ +# M8 Local Web Console Design + +## Goal + +Add a local, single-user Web console that makes the existing Mini CodeAgent runtime observable +and controllable without weakening its Workspace, Policy, approval, or Provider boundaries. + +## Product Scope + +The first screen is the working Agent console, not a landing page. It provides: + +- fixed startup Workspace and Provider status; +- a task transcript and prompt composer; +- live Agent lifecycle and Tool activity; +- explicit write/command approval with resources, argv, reason, risk, and diff; +- run cancellation; +- bounded run summaries and public errors; +- responsive desktop and mobile layouts. + +M8 does not add account management, cloud hosting, multi-user access, browser-side API-key +storage, durable conversation memory, image generation, arbitrary Workspace selection, or remote +network binding. + +## Visual Direction + +The console uses the recommended three-pane engineering workbench: + +- compact top bar for product, Workspace, Provider, model, connection state, and token usage; +- left rail for local sessions and Workspace context; +- center transcript for tasks, Agent output, progress, and prompt composition; +- right inspector for Activity and Changes, including approval actions and bounded diffs. + +The visual system is quiet and operational: neutral surfaces, charcoal text, teal active state, +amber pending state, green additions, and red deletions. It avoids gradients, decorative cards, +oversized typography, nested cards, and marketing composition. Corners are at most 6px. Dense +panels use 13-15px text and stable responsive tracks. + +## Architecture + +### Backend + +FastAPI is an explicit runtime dependency. `mini-code-agent web` starts Uvicorn on +`127.0.0.1` only and serves package-owned static assets. + +`WebRunManager` owns at most one active run. It creates: + +- a `WebApprovalHandler` that publishes a typed approval event and waits on a Future; +- a `WebEventSink` that serializes existing `AgentEvent` values; +- one background task that invokes the existing `run_task`; +- a bounded subscriber queue for Server-Sent Events. + +The manager publishes normalized envelopes: + +```json +{ + "sequence": 1, + "type": "run_started", + "payload": {} +} +``` + +It never publishes prompts, Tool arguments/results, API keys, or raw exception text through +lifecycle events. The explicit approval payload contains only the already bounded +`ApprovalRequest` preview needed for user authorization. + +### API + +- `GET /api/bootstrap`: product version, Workspace display path, Provider, model, key-configured + flag, and CSRF token. +- `POST /api/runs`: validate and start one task. +- `GET /api/runs/{run_id}/events`: replay retained events, then stream new SSE events. +- `POST /api/runs/{run_id}/approvals/{tool_call_id}`: approve or reject one pending action. +- `POST /api/runs/{run_id}/cancel`: cancel and join the active Agent task. +- `GET /healthz`: local process health. + +Requests that mutate state require an `X-Mini-Code-Agent-Token` header matching the random token +embedded into the bootstrap payload. + +### Frontend + +The frontend is package-owned HTML, CSS, and browser-native JavaScript. It has no CDN or remote +asset dependency and uses a small set of local Lucide-compatible SVG icons. + +The browser: + +- fetches bootstrap state; +- starts a run; +- subscribes to SSE; +- renders lifecycle rows and final output; +- opens approval content in the right inspector; +- posts approve/reject/cancel decisions with the CSRF token; +- reconnects with the last received sequence; +- disables incompatible controls while a run is active. + +## Security + +- `web` rejects non-loopback hosts instead of exposing the Agent over LAN. +- The Workspace is fixed by the CLI process and never accepted from browser input. +- API keys remain server-side settings. The browser receives only `api_key_configured`. +- Mutating routes require the process-random token. +- CORS is not enabled. Origin checks accept only the current loopback origin. +- Static files are selected from a fixed resource map, not user paths. +- Browser rendering uses `textContent`; model text, paths, reasons, commands, and diffs are never + inserted as HTML. +- Approvals are single-use and bound to run ID plus ToolCall ID. +- Disconnecting the browser does not approve anything. Pending approvals remain blocked until + explicit rejection, approval, cancellation, or process shutdown. +- The Web console remains local process governance, not an OS sandbox. + +## Error Handling + +Configuration failures are returned before a run starts. Concurrent start attempts return +HTTP 409. Unknown/stale approval IDs return 404 or 409 without executing a Tool. Queue overflow +cancels the run and emits a bounded terminal error. Client disconnects do not cancel the run; +the user can reconnect and replay retained events. + +## Testing + +- Unit tests cover manager lifecycle, approval, rejection, cancellation, stale decisions, event + redaction, and queue bounds with `ScriptedProvider`. +- ASGI tests cover bootstrap, CSRF, loopback origin checks, start/conflict, SSE replay, approval, + cancellation, health, and static assets. +- CLI tests cover loopback enforcement, option forwarding, and no-browser mode. +- Browser verification checks 1440x1024, 1024x768, and 390x844 for nonblank rendering, + stable layout, no overlap, working tabs, run state, approval state, and responsive collapse. +- No live Provider credential is required in CI. + +## SiliconFlow Compatibility + +The Web console reuses the M7 `OpenAICompatibleProvider`. SiliconFlow documents +`POST /v1/chat/completions`, Bearer authentication, SSE streaming, and function tools. M8 does +not call `POST /v1/images/generations`; that API is reserved for a later governed image Tool. + +The user-provided key is never committed. A live smoke requires +`MINI_CODE_AGENT_OPENAI_API_KEY` to be set in the process environment before launching the Web +console. diff --git a/pyproject.toml b/pyproject.toml index 2bf93e1..0df983f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "mini-code-agent" -version = "0.17.0a0" +version = "0.18.0a0" description = "A framework-light, provider-neutral, enterprise-grade mini code agent." readme = "README.md" requires-python = ">=3.12,<3.14" @@ -20,6 +20,7 @@ classifiers = [ ] dependencies = [ "defusedxml>=0.7.1,<0.8", + "fastapi>=0.116,<1", "httpx>=0.28,<1", "httpx-sse>=0.4,<1", "jsonschema>=4.23,<5", @@ -30,6 +31,7 @@ dependencies = [ "pyyaml>=6.0.2,<7", "rich>=13.9,<15", "typer>=0.15,<1", + "uvicorn>=0.34,<1", ] [project.scripts] diff --git a/src/mini_code_agent/cli.py b/src/mini_code_agent/cli.py index e929d80..6372c58 100644 --- a/src/mini_code_agent/cli.py +++ b/src/mini_code_agent/cli.py @@ -2,10 +2,13 @@ import asyncio import json +import threading +import webbrowser from pathlib import Path from typing import Annotated import typer +import uvicorn from rich.console import Console from rich.table import Table from rich.text import Text @@ -27,6 +30,7 @@ from mini_code_agent.logging import configure_logging from mini_code_agent.policy.models import SessionMode from mini_code_agent.terminal import TerminalApprovalHandler, TerminalEventSink +from mini_code_agent.web.app import create_web_app app = typer.Typer( name="mini-code-agent", @@ -36,6 +40,7 @@ ) console = Console() error_console = Console(stderr=True) +_LOOPBACK_HOSTS = frozenset({"127.0.0.1", "localhost", "::1"}) def _version_callback(value: bool) -> None: @@ -104,6 +109,12 @@ def _render_agent_result(result: AgentResult) -> None: error_console.print(f"[red]Agent stopped:[/red] {result.error}") +def _schedule_browser_open(url: str) -> None: + timer = threading.Timer(0.8, webbrowser.open, args=(url,)) + timer.daemon = True + timer.start() + + def _execute_task( settings: AppSettings, *, @@ -249,3 +260,53 @@ def chat( failed = failed or not result.succeeded if failed: raise typer.Exit(code=1) + + +@app.command() +def web( + workspace: Annotated[ + Path | None, + typer.Option("--workspace", "-w", help="Workspace directory. Defaults to the current one."), + ] = None, + config: Annotated[ + Path | None, + typer.Option("--config", help="Path to a TOML configuration file."), + ] = None, + host: Annotated[ + str, + typer.Option("--host", help="Loopback host for the local Web console."), + ] = "127.0.0.1", + port: Annotated[ + int, + typer.Option("--port", min=1, max=65535, help="Local Web console port."), + ] = 8765, + no_open: Annotated[ + bool, + typer.Option("--no-open", help="Do not open the system browser automatically."), + ] = False, +) -> None: + if host not in _LOOPBACK_HOSTS: + error_console.print( + "[red]Configuration error:[/red] Web console host must be a loopback address." + ) + raise typer.Exit(code=2) + + settings = _load_command_settings(config) + active_workspace = workspace or Path.cwd() + try: + web_app = create_web_app(settings, workspace=active_workspace) + except ValueError as exc: + error_console.print(f"[red]Configuration error:[/red] {exc}") + raise typer.Exit(code=2) from exc + + display_host = f"[{host}]" if host == "::1" else host + url = f"http://{display_host}:{port}" + console.print(f"[dim]Mini CodeAgent Web console: {url}[/dim]") + if not no_open: + _schedule_browser_open(url) + uvicorn.run( + web_app, + host=host, + port=port, + log_level=settings.log_level.value, + ) diff --git a/src/mini_code_agent/command/environment.py b/src/mini_code_agent/command/environment.py index 23b60da..09c7766 100644 --- a/src/mini_code_agent/command/environment.py +++ b/src/mini_code_agent/command/environment.py @@ -27,6 +27,7 @@ "LOCALAPPDATA", "PATH", "PATHEXT", + "SYSTEMDRIVE", "SYSTEMROOT", "TEMP", "TMP", diff --git a/src/mini_code_agent/web/__init__.py b/src/mini_code_agent/web/__init__.py new file mode 100644 index 0000000..4a84abf --- /dev/null +++ b/src/mini_code_agent/web/__init__.py @@ -0,0 +1 @@ +"""Local Web console adapters for Mini CodeAgent.""" diff --git a/src/mini_code_agent/web/app.py b/src/mini_code_agent/web/app.py new file mode 100644 index 0000000..5f5fc2f --- /dev/null +++ b/src/mini_code_agent/web/app.py @@ -0,0 +1,248 @@ +from __future__ import annotations + +import json +import secrets +from collections.abc import AsyncIterator, Awaitable, Callable +from importlib import resources +from pathlib import Path +from urllib.parse import urlsplit + +from fastapi import Depends, FastAPI, Header, HTTPException, Query, Request, status +from fastapi.responses import HTMLResponse, JSONResponse, Response, StreamingResponse + +from mini_code_agent import __version__ +from mini_code_agent.agent.events import EventSink +from mini_code_agent.agent.models import AgentResult +from mini_code_agent.application import run_task +from mini_code_agent.config import AppSettings, ProviderName +from mini_code_agent.policy.approval import ApprovalHandler +from mini_code_agent.policy.models import SessionMode +from mini_code_agent.web.manager import ( + RunConflictError, + RunNotFoundError, + WebRunManager, +) +from mini_code_agent.web.models import ( + ApprovalDecisionRequest, + RunSnapshot, + StartRunRequest, +) + +_LOOPBACK_HOSTS = frozenset({"127.0.0.1", "localhost", "::1"}) +_STATIC_ROOT = resources.files("mini_code_agent.web").joinpath("static") + + +def _is_loopback_origin(origin: str) -> bool: + try: + parsed = urlsplit(origin) + return parsed.scheme in {"http", "https"} and parsed.hostname in _LOOPBACK_HOSTS + except ValueError: + return False + + +def create_web_app( + settings: AppSettings, + *, + workspace: Path, + manager: WebRunManager | None = None, + csrf_token: str | None = None, +) -> FastAPI: + workspace_root = workspace.resolve() + if not workspace_root.is_dir(): + raise ValueError("Workspace must be an existing local directory.") + token = csrf_token or secrets.token_urlsafe(32) + + async def task_runner( + prompt: str, + approval: ApprovalHandler, + events: EventSink, + ) -> AgentResult: + return await run_task( + settings, + workspace=workspace_root, + user_prompt=prompt, + approval=approval, + session_mode=SessionMode.INTERACTIVE, + events=events, + ) + + run_manager = manager or WebRunManager(task_runner) + app = FastAPI( + title="Mini CodeAgent Web Console", + version=__version__, + docs_url=None, + redoc_url=None, + openapi_url=None, + ) + app.state.run_manager = run_manager + + async def enforce_local_origin( + request: Request, + call_next: Callable[[Request], Awaitable[Response]], + ) -> Response: + if request.method not in {"GET", "HEAD", "OPTIONS"}: + origin = request.headers.get("origin") + if origin is not None and not _is_loopback_origin(origin): + return _forbidden_response() + response: Response = await call_next(request) + response.headers["X-Content-Type-Options"] = "nosniff" + response.headers["Referrer-Policy"] = "no-referrer" + response.headers["Content-Security-Policy"] = ( + "default-src 'self'; script-src 'self'; style-src 'self'; " + "connect-src 'self'; img-src 'self' data:; frame-ancestors 'none'" + ) + return response + + app.middleware("http")(enforce_local_origin) + + def require_token( + x_mini_code_agent_token: str | None = Header(default=None), + ) -> None: + if x_mini_code_agent_token is None or not secrets.compare_digest( + x_mini_code_agent_token, + token, + ): + raise HTTPException( + status_code=status.HTTP_403_FORBIDDEN, + detail="Invalid local request token.", + ) + + async def health() -> dict[str, str]: + return {"status": "ok"} + + async def index() -> str: + return _static_text("index.html") + + async def styles() -> Response: + return Response(_static_text("styles.css"), media_type="text/css") + + async def javascript() -> Response: + return Response( + _static_text("app.js"), + media_type="text/javascript", + ) + + async def bootstrap() -> dict[str, object]: + key_configured = ( + settings.openai_api_key is not None + if settings.provider is ProviderName.OPENAI_COMPATIBLE + else settings.anthropic_api_key is not None + ) + active = run_manager.active_snapshot() + return { + "version": __version__, + "workspace": str(workspace_root), + "provider": settings.provider.value, + "model": settings.model, + "api_key_configured": key_configured, + "csrf_token": token, + "active_run": active.model_dump(mode="json") if active else None, + } + + async def start_run(payload: StartRunRequest) -> RunSnapshot: + try: + return await run_manager.start(payload.prompt) + except RunConflictError as exc: + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, + detail=str(exc), + ) from None + + async def run_events( + run_id: str, + after: int = Query(default=0, ge=0), + ) -> StreamingResponse: + try: + run_manager.snapshot(run_id) + except RunNotFoundError: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail="Run not found.", + ) from None + + async def stream() -> AsyncIterator[str]: + async for event in run_manager.subscribe( + run_id, + after_sequence=after, + ): + data = json.dumps(event.model_dump(mode="json"), separators=(",", ":")) + yield f"id: {event.sequence}\nevent: {event.type}\ndata: {data}\n\n" + + return StreamingResponse( + stream(), + media_type="text/event-stream", + headers={ + "Cache-Control": "no-cache, no-store", + "X-Accel-Buffering": "no", + }, + ) + + async def decide_approval( + run_id: str, + tool_call_id: str, + payload: ApprovalDecisionRequest, + ) -> dict[str, bool]: + accepted = await run_manager.decide_approval( + run_id, + tool_call_id, + payload.approved, + ) + if not accepted: + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, + detail="Approval is stale or no longer pending.", + ) + return {"accepted": True} + + async def cancel_run(run_id: str) -> dict[str, bool]: + cancelled = await run_manager.cancel(run_id) + if not cancelled: + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, + detail="Run is not active.", + ) + return {"cancelled": True} + + mutation_dependencies = [Depends(require_token)] + app.add_api_route("/healthz", health, methods=["GET"]) + app.add_api_route("/", index, methods=["GET"], response_class=HTMLResponse) + app.add_api_route("/static/styles.css", styles, methods=["GET"]) + app.add_api_route("/static/app.js", javascript, methods=["GET"]) + app.add_api_route("/api/bootstrap", bootstrap, methods=["GET"]) + app.add_api_route( + "/api/runs", + start_run, + methods=["POST"], + response_model=RunSnapshot, + status_code=status.HTTP_202_ACCEPTED, + dependencies=mutation_dependencies, + ) + app.add_api_route( + "/api/runs/{run_id}/events", + run_events, + methods=["GET"], + ) + app.add_api_route( + "/api/runs/{run_id}/approvals/{tool_call_id}", + decide_approval, + methods=["POST"], + dependencies=mutation_dependencies, + ) + app.add_api_route( + "/api/runs/{run_id}/cancel", + cancel_run, + methods=["POST"], + dependencies=mutation_dependencies, + ) + return app + + +def _forbidden_response() -> JSONResponse: + return JSONResponse( + status_code=status.HTTP_403_FORBIDDEN, + content={"detail": "Only loopback browser origins are allowed."}, + ) + + +def _static_text(name: str) -> str: + return _STATIC_ROOT.joinpath(name).read_text(encoding="utf-8") diff --git a/src/mini_code_agent/web/manager.py b/src/mini_code_agent/web/manager.py new file mode 100644 index 0000000..fb2312a --- /dev/null +++ b/src/mini_code_agent/web/manager.py @@ -0,0 +1,283 @@ +from __future__ import annotations + +import asyncio +from collections import deque +from collections.abc import AsyncIterator, Awaitable, Callable +from contextlib import suppress +from dataclasses import dataclass, field +from uuid import uuid4 + +from mini_code_agent.agent.events import AgentEvent, EventSink +from mini_code_agent.agent.models import AgentResult +from mini_code_agent.policy.approval import ApprovalHandler +from mini_code_agent.policy.models import ApprovalRequest +from mini_code_agent.web.models import RunSnapshot, WebEvent, WebRunStatus + +type TaskRunner = Callable[ + [str, ApprovalHandler, EventSink], + Awaitable[AgentResult], +] + + +class RunConflictError(RuntimeError): + """Raised when a second run is started while one is active.""" + + +class RunNotFoundError(KeyError): + """Raised when a Web run ID is unknown.""" + + +@dataclass +class _RunState: + run_id: str + status: WebRunStatus = WebRunStatus.RUNNING + events: deque[WebEvent] = field(default_factory=lambda: deque[WebEvent]()) + pending: dict[str, asyncio.Future[bool]] = field( + default_factory=lambda: dict[str, asyncio.Future[bool]]() + ) + subscribers: set[asyncio.Queue[None]] = field( + default_factory=lambda: set[asyncio.Queue[None]]() + ) + task: asyncio.Task[None] | None = None + next_sequence: int = 1 + + +class _WebEventSink: + def __init__(self, manager: WebRunManager, run_id: str) -> None: + self._manager = manager + self._run_id = run_id + + def publish(self, event: AgentEvent) -> None: + self._manager.publish_agent_event(self._run_id, event) + + +class _WebApprovalHandler: + def __init__(self, manager: WebRunManager, run_id: str) -> None: + self._manager = manager + self._run_id = run_id + + async def approve(self, request: ApprovalRequest) -> bool: + return await self._manager.request_approval(self._run_id, request) + + +class WebRunManager: + def __init__( + self, + runner: TaskRunner, + *, + max_retained_events: int = 512, + ) -> None: + if max_retained_events < 8: + raise ValueError("max_retained_events must be at least 8") + self._runner = runner + self._max_retained_events = max_retained_events + self._runs: dict[str, _RunState] = {} + self._active_run_id: str | None = None + + async def start(self, prompt: str) -> RunSnapshot: + if self._active_run_id is not None: + active = self._runs[self._active_run_id] + if active.status is WebRunStatus.RUNNING: + raise RunConflictError("A run is already active.") + + run_id = uuid4().hex + state = _RunState( + run_id=run_id, + events=deque(maxlen=self._max_retained_events), + ) + self._runs[run_id] = state + self._active_run_id = run_id + self._publish(run_id, "web_run_started", {"status": "running"}) + state.task = asyncio.create_task( + self._execute(run_id, prompt), + name=f"mini-code-agent-web-{run_id}", + ) + return self.snapshot(run_id) + + async def _execute(self, run_id: str, prompt: str) -> None: + state = self._state(run_id) + try: + result = await self._runner( + prompt, + _WebApprovalHandler(self, run_id), + _WebEventSink(self, run_id), + ) + except asyncio.CancelledError: + if state.status is WebRunStatus.RUNNING: + state.status = WebRunStatus.CANCELLED + self._publish(run_id, "web_run_cancelled", {"status": "cancelled"}) + except Exception: + state.status = WebRunStatus.FAILED + self._publish( + run_id, + "web_run_failed", + { + "status": "failed", + "message": "The Agent run failed. Check the local server logs.", + }, + ) + else: + state.status = WebRunStatus.COMPLETED + self._publish( + run_id, + "web_run_completed", + { + "status": "completed", + "stop_reason": result.stop_reason.value, + "turns": result.turns, + "tool_calls": result.tool_calls, + "usage": result.usage.model_dump(mode="json"), + "final_text": result.final_text, + "error": result.error, + }, + ) + finally: + self._reject_pending(state) + if self._active_run_id == run_id: + self._active_run_id = None + + async def decide_approval( + self, + run_id: str, + tool_call_id: str, + approved: bool, + ) -> bool: + state = self._runs.get(run_id) + if state is None or state.status is not WebRunStatus.RUNNING: + return False + future = state.pending.pop(tool_call_id, None) + if future is None or future.done(): + return False + future.set_result(approved) + self._publish( + run_id, + "approval_resolved", + {"tool_call_id": tool_call_id, "approved": approved}, + ) + return True + + def publish_agent_event(self, run_id: str, event: AgentEvent) -> None: + self._publish( + run_id, + "agent_event", + {"event": event.model_dump(mode="json")}, + ) + + async def request_approval( + self, + run_id: str, + request: ApprovalRequest, + ) -> bool: + state = self._state(run_id) + tool_call_id = request.preview.tool_call_id + if tool_call_id in state.pending: + return False + + future = asyncio.get_running_loop().create_future() + state.pending[tool_call_id] = future + self._publish( + run_id, + "approval_required", + request.model_dump(mode="json"), + ) + try: + return await future + finally: + state.pending.pop(tool_call_id, None) + + async def cancel(self, run_id: str) -> bool: + state = self._runs.get(run_id) + if state is None or state.status is not WebRunStatus.RUNNING: + return False + state.status = WebRunStatus.CANCELLED + self._reject_pending(state) + if state.task is not None and not state.task.done(): + state.task.cancel() + with suppress(asyncio.CancelledError): + await state.task + self._publish(run_id, "web_run_cancelled", {"status": "cancelled"}) + if self._active_run_id == run_id: + self._active_run_id = None + return True + + async def wait(self, run_id: str) -> RunSnapshot: + state = self._state(run_id) + if state.task is not None: + with suppress(asyncio.CancelledError): + await state.task + return self.snapshot(run_id) + + def snapshot(self, run_id: str) -> RunSnapshot: + state = self._state(run_id) + return RunSnapshot( + run_id=state.run_id, + status=state.status, + last_sequence=state.next_sequence - 1, + ) + + def active_snapshot(self) -> RunSnapshot | None: + if self._active_run_id is None: + return None + return self.snapshot(self._active_run_id) + + def events_after( + self, + run_id: str, + sequence: int = 0, + ) -> tuple[WebEvent, ...]: + state = self._state(run_id) + return tuple(event for event in state.events if event.sequence > sequence) + + async def subscribe( + self, + run_id: str, + *, + after_sequence: int = 0, + ) -> AsyncIterator[WebEvent]: + state = self._state(run_id) + wakeup: asyncio.Queue[None] = asyncio.Queue(maxsize=1) + state.subscribers.add(wakeup) + sequence = after_sequence + try: + while True: + pending = self.events_after(run_id, sequence) + for event in pending: + sequence = event.sequence + yield event + if state.status is not WebRunStatus.RUNNING: + return + await wakeup.get() + finally: + state.subscribers.discard(wakeup) + + def _publish( + self, + run_id: str, + event_type: str, + payload: dict[str, object], + ) -> None: + state = self._state(run_id) + event = WebEvent( + sequence=state.next_sequence, + type=event_type, + payload=payload, + ) + state.next_sequence += 1 + state.events.append(event) + for subscriber in tuple(state.subscribers): + if subscriber.empty(): + subscriber.put_nowait(None) + + def _state(self, run_id: str) -> _RunState: + try: + return self._runs[run_id] + except KeyError: + raise RunNotFoundError(run_id) from None + + @staticmethod + def _reject_pending(state: _RunState) -> None: + pending = tuple(state.pending.values()) + state.pending.clear() + for future in pending: + if not future.done(): + future.set_result(False) diff --git a/src/mini_code_agent/web/models.py b/src/mini_code_agent/web/models.py new file mode 100644 index 0000000..827595f --- /dev/null +++ b/src/mini_code_agent/web/models.py @@ -0,0 +1,39 @@ +from __future__ import annotations + +from enum import StrEnum +from typing import Any + +from pydantic import BaseModel, ConfigDict, Field + +_IDENTIFIER_PATTERN = r"^[A-Za-z0-9][A-Za-z0-9._-]{0,95}$" + + +class WebModel(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + +class StartRunRequest(WebModel): + prompt: str = Field(min_length=1, max_length=20_000) + + +class ApprovalDecisionRequest(WebModel): + approved: bool + + +class WebRunStatus(StrEnum): + RUNNING = "running" + COMPLETED = "completed" + CANCELLED = "cancelled" + FAILED = "failed" + + +class WebEvent(WebModel): + sequence: int = Field(ge=1) + type: str = Field(pattern=r"^[a-z][a-z0-9_]{0,63}$") + payload: dict[str, Any] = Field(default_factory=dict) + + +class RunSnapshot(WebModel): + run_id: str = Field(pattern=_IDENTIFIER_PATTERN) + status: WebRunStatus + last_sequence: int = Field(default=0, ge=0) diff --git a/src/mini_code_agent/web/static/app.js b/src/mini_code_agent/web/static/app.js new file mode 100644 index 0000000..ed782a1 --- /dev/null +++ b/src/mini_code_agent/web/static/app.js @@ -0,0 +1,442 @@ +"use strict"; + +const state = { + bootstrap: null, + runId: null, + lastSequence: 0, + eventSource: null, + pendingApproval: null, + activityCount: 0, +}; + +const elements = { + version: document.querySelector("#version-label"), + providerStatus: document.querySelector("#provider-status"), + provider: document.querySelector("#provider-label"), + model: document.querySelector("#model-label"), + workspace: document.querySelector("#workspace-label"), + sessionState: document.querySelector("#session-state"), + runState: document.querySelector("#run-state"), + transcript: document.querySelector("#transcript"), + emptyState: document.querySelector("#empty-state"), + composer: document.querySelector("#composer"), + prompt: document.querySelector("#prompt-input"), + runButton: document.querySelector("#run-button"), + cancelButton: document.querySelector("#cancel-button"), + activityList: document.querySelector("#activity-list"), + activityEmpty: document.querySelector("#activity-empty"), + activityCount: document.querySelector("#activity-count"), + activityTab: document.querySelector("#activity-tab"), + activityPanel: document.querySelector("#activity-panel"), + changesTab: document.querySelector("#changes-tab"), + changesPanel: document.querySelector("#changes-panel"), + approvalPanel: document.querySelector("#approval-panel"), + approvalRisk: document.querySelector("#approval-risk"), + approvalTitle: document.querySelector("#approval-title"), + approvalSummary: document.querySelector("#approval-summary"), + approvalReason: document.querySelector("#approval-reason"), + approvalResources: document.querySelector("#approval-resources"), + approvalCommandRow: document.querySelector("#approval-command-row"), + approvalCommand: document.querySelector("#approval-command"), + approveButton: document.querySelector("#approve-button"), + rejectButton: document.querySelector("#reject-button"), + changesEmpty: document.querySelector("#changes-empty"), + diffViewer: document.querySelector("#diff-viewer"), + inspector: document.querySelector("#inspector"), + inspectorToggle: document.querySelector("#inspector-toggle"), + inspectorClose: document.querySelector("#inspector-close"), + toastRegion: document.querySelector("#toast-region"), +}; + +function setText(element, value, fallback = "") { + element.textContent = value === null || value === undefined ? fallback : String(value); +} + +function setRunState(status, label) { + elements.runState.dataset.state = status; + setText(elements.runState, label); + const running = status === "running"; + elements.prompt.disabled = running; + elements.runButton.disabled = running || !state.bootstrap?.api_key_configured; + elements.cancelButton.disabled = !running; + setText(elements.sessionState, label); +} + +function showToast(message) { + const toast = document.createElement("div"); + toast.className = "toast"; + toast.textContent = message; + elements.toastRegion.append(toast); + window.setTimeout(() => toast.remove(), 4500); +} + +function addMessage(role, text, isError = false) { + elements.emptyState.hidden = true; + const article = document.createElement("article"); + article.className = `message ${role}${isError ? " error" : ""}`; + + const label = document.createElement("div"); + label.className = "message-label"; + const marker = document.createElement("span"); + marker.setAttribute("aria-hidden", "true"); + const labelText = document.createElement("strong"); + labelText.textContent = role === "user" ? "你" : "Agent"; + label.append(marker, labelText); + + const body = document.createElement("pre"); + body.className = "message-body"; + body.textContent = text; + article.append(label, body); + elements.transcript.append(article); + elements.transcript.scrollTop = elements.transcript.scrollHeight; +} + +function addActivity(title, detail, tone = "") { + elements.activityEmpty.hidden = true; + const item = document.createElement("li"); + item.className = `activity-item ${tone}`.trim(); + + const symbol = document.createElement("span"); + symbol.className = "activity-symbol"; + symbol.textContent = tone === "success" ? "✓" : tone === "error" ? "!" : "•"; + + const copy = document.createElement("div"); + copy.className = "activity-copy"; + const heading = document.createElement("strong"); + heading.textContent = title; + const description = document.createElement("small"); + description.textContent = detail; + copy.append(heading, description); + item.append(symbol, copy); + elements.activityList.append(item); + + state.activityCount += 1; + setText(elements.activityCount, state.activityCount); + elements.activityPanel.scrollTop = elements.activityPanel.scrollHeight; +} + +function showTab(name) { + const activity = name === "activity"; + elements.activityTab.classList.toggle("active", activity); + elements.activityTab.setAttribute("aria-selected", String(activity)); + elements.activityPanel.hidden = !activity; + elements.changesTab.classList.toggle("active", !activity); + elements.changesTab.setAttribute("aria-selected", String(!activity)); + elements.changesPanel.hidden = activity; +} + +function showApproval(payload) { + const preview = payload.preview; + state.pendingApproval = preview.tool_call_id; + elements.approvalPanel.hidden = false; + setText(elements.approvalRisk, preview.risk); + setText(elements.approvalTitle, preview.tool_name); + setText(elements.approvalSummary, preview.summary); + setText(elements.approvalReason, preview.reason); + setText(elements.approvalResources, (preview.resources || []).join("\n"), "无"); + const command = preview.command || []; + elements.approvalCommandRow.hidden = command.length === 0; + setText(elements.approvalCommand, command.join(" ")); + elements.approveButton.disabled = false; + elements.rejectButton.disabled = false; + + const diff = preview.diff || ""; + elements.changesEmpty.hidden = diff.length > 0; + elements.diffViewer.hidden = diff.length === 0; + setText(elements.diffViewer, diff); + addActivity("等待操作审批", preview.summary, "pending"); + elements.inspector.classList.add("open"); + elements.inspectorToggle.setAttribute("aria-expanded", "true"); +} + +function clearApproval() { + state.pendingApproval = null; + elements.approvalPanel.hidden = true; + elements.approveButton.disabled = false; + elements.rejectButton.disabled = false; +} + +function describeAgentEvent(event) { + const type = event.type; + if (type === "run_started") { + return ["Agent 已启动", `最多 ${event.max_turns} 轮`]; + } + if (type === "model_started") { + return ["模型正在思考", `第 ${event.turn} 轮`]; + } + if (type === "model_completed") { + const usage = event.usage || {}; + return [ + "模型响应完成", + `${event.finish_reason} · 输入 ${usage.input_tokens || 0} / 输出 ${usage.output_tokens || 0}`, + ]; + } + if (type === "tool_started") { + return [`调用工具 ${event.tool_name}`, `${event.side_effect} · 第 ${event.turn} 轮`]; + } + if (type === "tool_completed") { + return [ + `工具完成 ${event.tool_name}`, + event.is_error ? "执行返回错误" : "执行成功", + event.is_error ? "error" : "success", + ]; + } + if (type === "context_compacted") { + return ["上下文已压缩", `省略 ${event.omitted_messages} 条消息`]; + } + if (type === "run_stopped") { + return [ + "Agent 运行结束", + `${event.reason} · ${event.turns} 轮 · ${event.tool_calls} 次工具调用`, + event.error ? "error" : "success", + ]; + } + return [type, "Agent 生命周期事件"]; +} + +function stopEventStream() { + if (state.eventSource !== null) { + state.eventSource.close(); + state.eventSource = null; + } +} + +function handleEvent(envelope) { + state.lastSequence = Math.max(state.lastSequence, envelope.sequence); + if (envelope.type === "agent_event") { + const description = describeAgentEvent(envelope.payload.event); + addActivity(description[0], description[1], description[2] || ""); + return; + } + if (envelope.type === "approval_required") { + showApproval(envelope.payload); + return; + } + if (envelope.type === "approval_resolved") { + clearApproval(); + addActivity( + envelope.payload.approved ? "操作已允许" : "操作已拒绝", + envelope.payload.tool_call_id, + envelope.payload.approved ? "success" : "error", + ); + return; + } + if (envelope.type === "web_run_completed") { + clearApproval(); + const finalText = envelope.payload.final_text || envelope.payload.error || "运行已结束。"; + addMessage("assistant", finalText, Boolean(envelope.payload.error)); + addActivity( + "任务完成", + `${envelope.payload.turns} 轮 · ${envelope.payload.tool_calls} 次工具调用`, + envelope.payload.error ? "error" : "success", + ); + setRunState(envelope.payload.error ? "failed" : "completed", "已完成"); + stopEventStream(); + state.runId = null; + return; + } + if (envelope.type === "web_run_failed") { + clearApproval(); + addMessage("assistant", envelope.payload.message, true); + addActivity("任务失败", envelope.payload.message, "error"); + setRunState("failed", "失败"); + stopEventStream(); + state.runId = null; + return; + } + if (envelope.type === "web_run_cancelled") { + clearApproval(); + addActivity("任务已停止", "运行由用户取消", "error"); + setRunState("idle", "已停止"); + stopEventStream(); + state.runId = null; + } +} + +function connectEvents(runId) { + stopEventStream(); + const source = new EventSource( + `/api/runs/${encodeURIComponent(runId)}/events?after=${state.lastSequence}`, + ); + state.eventSource = source; + + const eventNames = [ + "web_run_started", + "agent_event", + "approval_required", + "approval_resolved", + "web_run_completed", + "web_run_failed", + "web_run_cancelled", + ]; + for (const eventName of eventNames) { + source.addEventListener(eventName, (event) => { + try { + handleEvent(JSON.parse(event.data)); + } catch { + showToast("收到无法解析的运行事件。"); + } + }); + } + source.onerror = () => { + if (state.runId !== null) { + addActivity("连接正在恢复", "等待本地事件流重新连接", "pending"); + } + }; +} + +async function api(path, options = {}) { + const headers = new Headers(options.headers || {}); + if (options.body !== undefined) { + headers.set("Content-Type", "application/json"); + } + if (options.method && options.method !== "GET") { + headers.set("X-Mini-Code-Agent-Token", state.bootstrap.csrf_token); + } + const response = await fetch(path, { ...options, headers }); + if (!response.ok) { + let message = `请求失败 (${response.status})`; + try { + const payload = await response.json(); + message = payload.detail || message; + } catch { + // Keep the bounded public fallback. + } + throw new Error(message); + } + return response.json(); +} + +async function startRun(prompt) { + setRunState("running", "运行中"); + state.lastSequence = 0; + state.activityCount = 0; + elements.activityList.replaceChildren(); + elements.activityEmpty.hidden = false; + setText(elements.activityCount, "0"); + addMessage("user", prompt); + addActivity("任务已提交", "正在初始化本地 Agent", "pending"); + try { + const snapshot = await api("/api/runs", { + method: "POST", + body: JSON.stringify({ prompt }), + }); + state.runId = snapshot.run_id; + connectEvents(snapshot.run_id); + } catch (error) { + setRunState("failed", "启动失败"); + addMessage("assistant", error.message, true); + showToast(error.message); + } +} + +async function decideApproval(approved) { + if (state.runId === null || state.pendingApproval === null) { + return; + } + elements.approveButton.disabled = true; + elements.rejectButton.disabled = true; + const toolCallId = state.pendingApproval; + try { + await api( + `/api/runs/${encodeURIComponent(state.runId)}/approvals/${encodeURIComponent(toolCallId)}`, + { + method: "POST", + body: JSON.stringify({ approved }), + }, + ); + } catch (error) { + showToast(error.message); + clearApproval(); + } +} + +async function cancelRun() { + if (state.runId === null) { + return; + } + elements.cancelButton.disabled = true; + try { + await api(`/api/runs/${encodeURIComponent(state.runId)}/cancel`, { + method: "POST", + }); + } catch (error) { + showToast(error.message); + elements.cancelButton.disabled = false; + } +} + +async function bootstrap() { + try { + const response = await fetch("/api/bootstrap", { cache: "no-store" }); + if (!response.ok) { + throw new Error("无法读取本地运行配置。"); + } + state.bootstrap = await response.json(); + setText(elements.version, `v${state.bootstrap.version}`); + setText(elements.provider, state.bootstrap.provider); + setText(elements.model, state.bootstrap.model, "未配置模型"); + setText(elements.workspace, state.bootstrap.workspace); + elements.providerStatus.classList.toggle( + "ready", + state.bootstrap.api_key_configured && Boolean(state.bootstrap.model), + ); + elements.providerStatus.classList.toggle( + "error", + !state.bootstrap.api_key_configured || !state.bootstrap.model, + ); + setText( + elements.provider, + state.bootstrap.api_key_configured ? state.bootstrap.provider : "服务端密钥未配置", + ); + elements.runButton.disabled = + !state.bootstrap.api_key_configured || !state.bootstrap.model; + if (!state.bootstrap.api_key_configured || !state.bootstrap.model) { + showToast("请在启动服务前配置模型和服务端环境变量。"); + } + if (state.bootstrap.active_run) { + state.runId = state.bootstrap.active_run.run_id; + setRunState("running", "运行中"); + connectEvents(state.runId); + } + } catch (error) { + elements.providerStatus.classList.add("error"); + setText(elements.provider, "本地服务不可用"); + elements.runButton.disabled = true; + showToast(error.message); + } +} + +elements.composer.addEventListener("submit", (event) => { + event.preventDefault(); + const prompt = elements.prompt.value.trim(); + if (prompt.length === 0 || state.runId !== null) { + return; + } + elements.prompt.value = ""; + startRun(prompt); +}); + +elements.prompt.addEventListener("keydown", (event) => { + if (event.key === "Enter" && event.ctrlKey) { + event.preventDefault(); + elements.composer.requestSubmit(); + } +}); + +elements.cancelButton.addEventListener("click", cancelRun); +elements.approveButton.addEventListener("click", () => decideApproval(true)); +elements.rejectButton.addEventListener("click", () => decideApproval(false)); +elements.activityTab.addEventListener("click", () => showTab("activity")); +elements.changesTab.addEventListener("click", () => showTab("changes")); +elements.inspectorToggle.addEventListener("click", () => { + const opened = elements.inspector.classList.toggle("open"); + elements.inspectorToggle.setAttribute("aria-expanded", String(opened)); +}); +elements.inspectorClose.addEventListener("click", () => { + elements.inspector.classList.remove("open"); + elements.inspectorToggle.setAttribute("aria-expanded", "false"); +}); + +window.addEventListener("beforeunload", stopEventStream); +bootstrap(); diff --git a/src/mini_code_agent/web/static/index.html b/src/mini_code_agent/web/static/index.html new file mode 100644 index 0000000..7e90917 --- /dev/null +++ b/src/mini_code_agent/web/static/index.html @@ -0,0 +1,207 @@ + + + + + + + Mini CodeAgent + + + + +
+
+ + Mini CodeAgent + local +
+
+ + + 正在读取配置 + + 未配置模型 +
+ +
+ +
+ + +
+
+
+ 本地 Agent 工作台 +

项目任务

+
+
空闲
+
+ +
+
+ +

从一个具体任务开始

+

例如:分析项目结构并说明核心调用链。

+
+
+ +
+ + + +
+
+ + +
+ +
+ + + diff --git a/src/mini_code_agent/web/static/styles.css b/src/mini_code_agent/web/static/styles.css new file mode 100644 index 0000000..28caf88 --- /dev/null +++ b/src/mini_code_agent/web/static/styles.css @@ -0,0 +1,958 @@ +:root { + --topbar-height: 52px; + --rail-width: 232px; + --inspector-width: 360px; + --bg: #f4f6f6; + --surface: #ffffff; + --surface-subtle: #f8f9f9; + --border: #d8dede; + --border-strong: #b9c3c3; + --text: #172222; + --text-muted: #637171; + --teal: #0c7069; + --teal-hover: #095e58; + --teal-soft: #e6f3f1; + --amber: #9a5b00; + --amber-soft: #fff3d9; + --green: #176b3a; + --green-soft: #e8f5ec; + --red: #a13535; + --red-soft: #fbecec; + --shadow: 0 8px 24px rgb(20 35 35 / 10%); + color: var(--text); + font-family: + Inter, "Segoe UI", "Microsoft YaHei", system-ui, -apple-system, sans-serif; + font-size: 14px; + letter-spacing: 0; +} + +* { + box-sizing: border-box; +} + +html, +body { + width: 100%; + height: 100%; + margin: 0; + overflow: hidden; + background: var(--bg); +} + +button, +textarea { + font: inherit; + letter-spacing: 0; +} + +button { + color: inherit; +} + +svg { + width: 18px; + height: 18px; + fill: none; + stroke: currentColor; + stroke-linecap: round; + stroke-linejoin: round; + stroke-width: 1.8; +} + +.sr-only { + position: absolute; + width: 1px; + height: 1px; + padding: 0; + margin: -1px; + overflow: hidden; + clip: rect(0, 0, 0, 0); + white-space: nowrap; + border: 0; +} + +.topbar { + position: relative; + z-index: 20; + height: var(--topbar-height); + display: flex; + align-items: center; + justify-content: space-between; + gap: 20px; + padding: 0 14px; + border-bottom: 1px solid var(--border); + background: var(--surface); +} + +.brand, +.runtime-status { + display: flex; + align-items: center; + min-width: 0; +} + +.brand { + gap: 9px; + font-weight: 680; +} + +.brand-mark { + display: grid; + width: 27px; + height: 27px; + place-items: center; + border-radius: 5px; + background: var(--text); + color: #fff; + font-size: 13px; +} + +.version { + color: var(--text-muted); + font-size: 11px; + font-weight: 550; +} + +.runtime-status { + justify-content: flex-end; + gap: 12px; + overflow: hidden; +} + +.status-chip { + display: inline-flex; + align-items: center; + gap: 7px; + min-width: 0; + color: var(--text-muted); + font-size: 12px; +} + +.status-dot { + flex: 0 0 auto; + width: 7px; + height: 7px; + border-radius: 50%; + background: var(--amber); +} + +.status-chip.ready .status-dot { + background: var(--green); +} + +.status-chip.error .status-dot { + background: var(--red); +} + +.model-label { + max-width: 270px; + padding-left: 12px; + overflow: hidden; + border-left: 1px solid var(--border); + color: var(--text-muted); + font-family: "Cascadia Code", Consolas, monospace; + font-size: 11px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.workbench { + height: calc(100% - var(--topbar-height)); + display: grid; + grid-template-columns: var(--rail-width) minmax(390px, 1fr) var(--inspector-width); + overflow: hidden; +} + +.session-rail, +.inspector { + min-width: 0; + background: var(--surface-subtle); +} + +.session-rail { + display: flex; + flex-direction: column; + padding: 14px 10px; + border-right: 1px solid var(--border); +} + +.rail-heading { + height: 32px; + display: flex; + align-items: center; + justify-content: space-between; + padding: 0 5px 8px; + color: var(--text-muted); + font-size: 11px; + font-weight: 700; + text-transform: uppercase; +} + +.icon-button { + display: inline-grid; + width: 30px; + height: 30px; + padding: 0; + place-items: center; + border: 1px solid transparent; + border-radius: 5px; + background: transparent; + cursor: pointer; +} + +.icon-button:hover:not(:disabled) { + border-color: var(--border); + background: var(--surface); +} + +.icon-button:disabled { + color: var(--border-strong); + cursor: default; +} + +.session-item { + width: 100%; + min-height: 54px; + display: flex; + align-items: center; + gap: 10px; + padding: 8px 9px; + border: 1px solid transparent; + border-radius: 6px; + background: transparent; + text-align: left; + cursor: pointer; +} + +.session-item.active { + border-color: #c5dcda; + background: var(--teal-soft); +} + +.session-icon { + display: grid; + flex: 0 0 auto; + width: 30px; + height: 30px; + place-items: center; + border-radius: 5px; + background: var(--surface); + color: var(--teal); +} + +.session-copy { + min-width: 0; + display: flex; + flex-direction: column; + gap: 3px; +} + +.session-copy strong { + font-size: 13px; +} + +.session-copy small { + color: var(--text-muted); + font-size: 11px; +} + +.workspace-block { + margin-top: 20px; + padding: 0 6px; +} + +.section-label, +.eyebrow { + display: block; + color: var(--text-muted); + font-size: 10px; + font-weight: 700; + text-transform: uppercase; +} + +.workspace-path { + display: flex; + align-items: flex-start; + gap: 7px; + margin-top: 9px; + color: var(--text-muted); + font-family: "Cascadia Code", Consolas, monospace; + font-size: 11px; + line-height: 1.45; +} + +.workspace-path svg { + flex: 0 0 auto; +} + +.workspace-path span { + min-width: 0; + overflow-wrap: anywhere; +} + +.rail-note { + display: flex; + align-items: flex-start; + gap: 8px; + margin-top: auto; + padding: 10px 7px 2px; + color: var(--text-muted); + font-size: 11px; + line-height: 1.45; +} + +.rail-note svg { + flex: 0 0 auto; + color: var(--teal); +} + +.conversation { + min-width: 0; + display: grid; + grid-template-rows: auto minmax(0, 1fr) auto; + background: var(--surface); +} + +.conversation-heading { + min-height: 70px; + display: flex; + align-items: center; + justify-content: space-between; + padding: 12px 22px; + border-bottom: 1px solid var(--border); +} + +.conversation-heading h1 { + margin: 4px 0 0; + font-size: 18px; + font-weight: 680; +} + +.run-state { + display: inline-flex; + align-items: center; + min-height: 25px; + padding: 0 9px; + border: 1px solid var(--border); + border-radius: 4px; + color: var(--text-muted); + font-size: 11px; +} + +.run-state[data-state="running"] { + border-color: #d6b46d; + background: var(--amber-soft); + color: var(--amber); +} + +.run-state[data-state="completed"] { + border-color: #a7d3b7; + background: var(--green-soft); + color: var(--green); +} + +.run-state[data-state="failed"] { + border-color: #e4b1b1; + background: var(--red-soft); + color: var(--red); +} + +.transcript { + min-height: 0; + overflow-y: auto; + padding: 24px clamp(18px, 5vw, 72px); + scroll-behavior: smooth; +} + +.empty-state { + height: 100%; + min-height: 240px; + display: flex; + flex-direction: column; + align-items: center; + justify-content: center; + color: var(--text-muted); + text-align: center; +} + +.empty-icon { + display: grid; + width: 42px; + height: 42px; + place-items: center; + border: 1px solid var(--border); + border-radius: 6px; + background: var(--surface-subtle); + color: var(--teal); +} + +.empty-state h2 { + margin: 14px 0 4px; + color: var(--text); + font-size: 15px; +} + +.empty-state p { + margin: 0; + font-size: 13px; +} + +.message { + max-width: 820px; + margin: 0 auto 24px; +} + +.message-label { + display: flex; + align-items: center; + gap: 8px; + margin-bottom: 7px; + color: var(--text-muted); + font-size: 11px; + font-weight: 700; + text-transform: uppercase; +} + +.message-label span { + width: 7px; + height: 7px; + border-radius: 50%; + background: var(--border-strong); +} + +.message.assistant .message-label span { + background: var(--teal); +} + +.message-body { + margin: 0; + color: var(--text); + font-family: inherit; + font-size: 14px; + line-height: 1.65; + overflow-wrap: anywhere; + white-space: pre-wrap; +} + +.message.user .message-body { + padding: 12px 14px; + border-left: 3px solid var(--border-strong); + background: var(--surface-subtle); +} + +.message.error .message-body { + color: var(--red); +} + +.composer { + padding: 12px clamp(16px, 4vw, 56px) 16px; + border-top: 1px solid var(--border); + background: var(--surface); +} + +.composer:focus-within { + background: #fcfdfd; +} + +#prompt-input { + width: 100%; + min-height: 70px; + max-height: 190px; + resize: vertical; + padding: 11px 12px; + border: 1px solid var(--border-strong); + border-radius: 6px; + outline: none; + color: var(--text); + background: var(--surface); + line-height: 1.5; +} + +#prompt-input:focus { + border-color: var(--teal); + box-shadow: 0 0 0 2px rgb(12 112 105 / 12%); +} + +#prompt-input:disabled { + background: var(--surface-subtle); + color: var(--text-muted); +} + +.composer-footer { + min-height: 38px; + display: flex; + align-items: flex-end; + justify-content: space-between; + gap: 12px; + padding-top: 9px; +} + +.composer-hint { + color: var(--text-muted); + font-size: 11px; +} + +.composer-actions, +.approval-actions { + display: flex; + align-items: center; + gap: 8px; +} + +.button { + min-height: 34px; + display: inline-flex; + align-items: center; + justify-content: center; + gap: 7px; + padding: 0 13px; + border: 1px solid transparent; + border-radius: 5px; + font-size: 12px; + font-weight: 650; + cursor: pointer; +} + +.button svg { + width: 15px; + height: 15px; +} + +.button.primary { + border-color: var(--teal); + background: var(--teal); + color: #fff; +} + +.button.primary:hover:not(:disabled) { + border-color: var(--teal-hover); + background: var(--teal-hover); +} + +.button.secondary { + border-color: var(--border-strong); + background: var(--surface); +} + +.button.danger { + border-color: #d4a1a1; + background: var(--surface); + color: var(--red); +} + +.button:disabled { + border-color: var(--border); + background: var(--surface-subtle); + color: #9ba6a6; + cursor: default; +} + +.inspector { + display: grid; + grid-template-rows: 45px minmax(0, 1fr); + border-left: 1px solid var(--border); +} + +.inspector-tabs { + display: flex; + align-items: stretch; + padding: 0 8px; + border-bottom: 1px solid var(--border); + background: var(--surface); +} + +.tab { + min-width: 82px; + display: flex; + align-items: center; + justify-content: center; + gap: 7px; + padding: 0 10px; + border: 0; + border-bottom: 2px solid transparent; + background: transparent; + color: var(--text-muted); + font-size: 12px; + font-weight: 650; + cursor: pointer; +} + +.tab.active { + border-bottom-color: var(--teal); + color: var(--text); +} + +.count { + min-width: 18px; + padding: 1px 5px; + border-radius: 8px; + background: var(--surface-subtle); + color: var(--text-muted); + font-size: 10px; + text-align: center; +} + +.inspector-close { + display: none; + margin: auto 0 auto auto; +} + +.inspector-panel { + min-height: 0; + overflow-y: auto; + padding: 14px; +} + +.panel-empty { + padding: 24px 12px; + color: var(--text-muted); + font-size: 12px; + line-height: 1.6; + text-align: center; +} + +.activity-list { + display: flex; + flex-direction: column; + gap: 0; + padding: 0; + margin: 0; + list-style: none; +} + +.activity-item { + position: relative; + display: grid; + grid-template-columns: 24px minmax(0, 1fr); + gap: 9px; + min-height: 51px; + padding: 7px 0; +} + +.activity-item:not(:last-child)::after { + position: absolute; + top: 31px; + bottom: -5px; + left: 11px; + width: 1px; + background: var(--border); + content: ""; +} + +.activity-symbol { + position: relative; + z-index: 1; + display: grid; + width: 24px; + height: 24px; + place-items: center; + border: 1px solid var(--border); + border-radius: 50%; + background: var(--surface); + color: var(--text-muted); + font-size: 10px; + font-weight: 700; +} + +.activity-item.pending .activity-symbol { + border-color: #d6b46d; + background: var(--amber-soft); + color: var(--amber); +} + +.activity-item.success .activity-symbol { + border-color: #a7d3b7; + background: var(--green-soft); + color: var(--green); +} + +.activity-item.error .activity-symbol { + border-color: #e4b1b1; + background: var(--red-soft); + color: var(--red); +} + +.activity-copy { + min-width: 0; + padding-top: 2px; +} + +.activity-copy strong, +.activity-copy small { + display: block; +} + +.activity-copy strong { + overflow: hidden; + font-size: 12px; + font-weight: 650; + text-overflow: ellipsis; + white-space: nowrap; +} + +.activity-copy small { + margin-top: 3px; + color: var(--text-muted); + font-size: 11px; + line-height: 1.35; + overflow-wrap: anywhere; +} + +.approval-panel { + margin-top: 14px; + padding: 13px; + border: 1px solid #dbbd7e; + border-radius: 6px; + background: #fffaf0; +} + +.approval-heading { + display: flex; + align-items: center; + gap: 8px; +} + +.approval-heading strong { + min-width: 0; + overflow: hidden; + font-size: 13px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.risk-badge { + flex: 0 0 auto; + padding: 2px 6px; + border-radius: 3px; + background: var(--amber-soft); + color: var(--amber); + font-size: 9px; + font-weight: 750; + text-transform: uppercase; +} + +.approval-panel > p { + margin: 10px 0 12px; + font-size: 12px; + line-height: 1.5; +} + +.approval-details { + margin: 0 0 13px; +} + +.approval-details div { + padding: 7px 0; + border-top: 1px solid #ead9b8; +} + +.approval-details dt { + margin-bottom: 4px; + color: var(--text-muted); + font-size: 10px; + font-weight: 700; + text-transform: uppercase; +} + +.approval-details dd { + margin: 0; + font-size: 11px; + line-height: 1.45; + overflow-wrap: anywhere; +} + +.approval-details code { + font-family: "Cascadia Code", Consolas, monospace; + font-size: 10px; +} + +.approval-actions { + justify-content: flex-end; +} + +.diff-viewer { + min-height: 160px; + padding: 12px; + margin: 0; + overflow: auto; + border: 1px solid var(--border); + border-radius: 5px; + background: #182020; + color: #e4ecec; + font-family: "Cascadia Code", Consolas, monospace; + font-size: 11px; + line-height: 1.55; + white-space: pre; +} + +.toast-region { + position: fixed; + z-index: 50; + right: 16px; + bottom: 16px; + display: flex; + flex-direction: column; + gap: 8px; + pointer-events: none; +} + +.toast { + max-width: min(360px, calc(100vw - 32px)); + padding: 10px 12px; + border: 1px solid var(--border-strong); + border-radius: 5px; + background: var(--surface); + box-shadow: var(--shadow); + color: var(--text); + font-size: 12px; + line-height: 1.4; +} + +.inspector-toggle { + display: none; +} + +@media (max-width: 1100px) { + :root { + --rail-width: 190px; + --inspector-width: 320px; + } + + .model-label { + max-width: 180px; + } +} + +@media (max-width: 880px) { + .workbench { + grid-template-columns: minmax(360px, 1fr) 300px; + } + + .session-rail { + display: none; + } + + .runtime-status { + max-width: 55%; + } +} + +@media (max-width: 720px) { + .topbar { + gap: 8px; + padding: 0 10px; + } + + .brand .version, + .model-label { + display: none; + } + + .runtime-status { + max-width: none; + margin-left: auto; + } + + .status-chip { + max-width: 132px; + } + + #provider-label { + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; + } + + .inspector-toggle { + display: inline-grid; + flex: 0 0 auto; + } + + .workbench { + display: block; + } + + .conversation { + height: 100%; + } + + .conversation-heading { + min-height: 60px; + padding: 10px 14px; + } + + .conversation-heading h1 { + font-size: 16px; + } + + .transcript { + padding: 18px 14px; + } + + .composer { + padding: 10px 10px 12px; + } + + .composer-hint { + display: none; + } + + .composer-footer { + justify-content: flex-end; + } + + .inspector { + position: fixed; + z-index: 40; + right: 0; + bottom: 0; + left: 0; + height: min(72vh, 620px); + display: grid; + transform: translateY(102%); + border-top: 1px solid var(--border-strong); + border-left: 0; + box-shadow: 0 -12px 36px rgb(20 35 35 / 16%); + transition: transform 180ms ease; + } + + .inspector.open { + transform: translateY(0); + } + + .inspector-close { + display: inline-grid; + } + + .inspector-panel { + padding: 12px; + } +} + +@media (max-width: 420px) { + .brand { + font-size: 13px; + } + + .status-chip { + max-width: 112px; + } + + #prompt-input { + min-height: 62px; + } + + .button { + min-height: 33px; + padding: 0 11px; + } +} + +@media (prefers-reduced-motion: reduce) { + *, + *::before, + *::after { + scroll-behavior: auto !important; + transition-duration: 0.01ms !important; + } +} diff --git a/tests/cli/test_cli.py b/tests/cli/test_cli.py index d96df64..ff1c63d 100644 --- a/tests/cli/test_cli.py +++ b/tests/cli/test_cli.py @@ -52,7 +52,7 @@ def test_version_option_prints_package_version() -> None: result = runner.invoke(app, ["--version"]) assert result.exit_code == 0 - assert result.stdout.strip() == "0.17.0a0" + assert result.stdout.strip() == "0.18.0a0" def test_module_entrypoint_prints_package_version() -> None: @@ -64,7 +64,7 @@ def test_module_entrypoint_prints_package_version() -> None: ) assert result.returncode == 0 - assert result.stdout.strip() == "0.17.0a0" + assert result.stdout.strip() == "0.18.0a0" def test_doctor_json_never_prints_secrets( @@ -260,3 +260,98 @@ async def fake_run_task(*args: object, **kwargs: object) -> AgentResult: assert "Completed: Inspect files." in result.stdout assert "Completed: Summarize changes." in result.stdout assert "independent bounded run" in result.stdout + + +def test_web_starts_loopback_server_and_opens_browser( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + fake_app = object() + created: list[dict[str, object]] = [] + served: list[dict[str, object]] = [] + opened: list[str] = [] + + def fake_create_web_app(*args: object, **kwargs: object) -> object: + created.append(dict(kwargs)) + return fake_app + + def fake_uvicorn_run(app_object: object, **kwargs: object) -> None: + assert app_object is fake_app + served.append(dict(kwargs)) + + monkeypatch.setattr("mini_code_agent.cli.create_web_app", fake_create_web_app) + monkeypatch.setattr("mini_code_agent.cli.uvicorn.run", fake_uvicorn_run) + + def fake_open(url: str) -> None: + opened.append(url) + + monkeypatch.setattr("mini_code_agent.cli._schedule_browser_open", fake_open) + monkeypatch.setenv("MINI_CODE_AGENT_MODEL", "Pro/zai-org/GLM-4.7") + monkeypatch.setenv("MINI_CODE_AGENT_OPENAI_API_KEY", "test-key") + + result = runner.invoke( + app, + [ + "web", + "--workspace", + str(tmp_path), + "--port", + "9876", + ], + ) + + assert result.exit_code == 0 + assert created[0]["workspace"] == tmp_path + assert served == [ + { + "host": "127.0.0.1", + "port": 9876, + "log_level": "info", + } + ] + assert opened == ["http://127.0.0.1:9876"] + + +def test_web_no_open_skips_browser( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + def fake_create_web_app(*args: object, **kwargs: object) -> object: + del args, kwargs + return object() + + def fake_uvicorn_run(*args: object, **kwargs: object) -> None: + del args, kwargs + + def fail_open(url: str) -> None: + pytest.fail(f"unexpected browser open: {url}") + + monkeypatch.setattr("mini_code_agent.cli.create_web_app", fake_create_web_app) + monkeypatch.setattr("mini_code_agent.cli.uvicorn.run", fake_uvicorn_run) + monkeypatch.setattr("mini_code_agent.cli._schedule_browser_open", fail_open) + + result = runner.invoke( + app, + ["web", "--workspace", str(tmp_path), "--no-open"], + ) + + assert result.exit_code == 0 + + +def test_web_rejects_non_loopback_host( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + def fail_uvicorn(*args: object, **kwargs: object) -> None: + del args, kwargs + pytest.fail("server must not start") + + monkeypatch.setattr("mini_code_agent.cli.uvicorn.run", fail_uvicorn) + + result = runner.invoke( + app, + ["web", "--workspace", str(tmp_path), "--host", "0.0.0.0"], + ) + + assert result.exit_code == 2 + assert "loopback" in result.stderr.lower() diff --git a/tests/integration/test_agent_loop.py b/tests/integration/test_agent_loop.py index 09b2c1e..15b818e 100644 --- a/tests/integration/test_agent_loop.py +++ b/tests/integration/test_agent_loop.py @@ -53,7 +53,7 @@ async def test_fake_provider_drives_native_tool_call_round_trip() -> None: assert tool_result_message.role is MessageRole.USER assert tool_result_message.tool_results[0].tool_call_id == "call-1" payload = json.loads(tool_result_message.tool_results[0].content) - assert payload["package_version"] == "0.17.0a0" + assert payload["package_version"] == "0.18.0a0" assert [type(event) for event in events.events] == [ RunStarted, ModelStarted, diff --git a/tests/unit/command/test_environment.py b/tests/unit/command/test_environment.py index a83ae1a..9067093 100644 --- a/tests/unit/command/test_environment.py +++ b/tests/unit/command/test_environment.py @@ -27,6 +27,7 @@ def test_windows_environment_lookup_is_case_insensitive_and_canonical() -> None: source = { "Path": r"C:\Windows\System32", "pathext": ".EXE;.CMD", + "systemdrive": "C:", "systemroot": r"C:\Windows", "TEMP": r"C:\Temp", "OPENAI_API_KEY": "secret", @@ -38,6 +39,7 @@ def test_windows_environment_lookup_is_case_insensitive_and_canonical() -> None: assert result == { "PATH": r"C:\Windows\System32", "PATHEXT": ".EXE;.CMD", + "SYSTEMDRIVE": "C:", "SYSTEMROOT": r"C:\Windows", "TEMP": r"C:\Temp", } diff --git a/tests/unit/test_package.py b/tests/unit/test_package.py index f39e737..9a64612 100644 --- a/tests/unit/test_package.py +++ b/tests/unit/test_package.py @@ -4,7 +4,7 @@ def test_package_exports_release_version() -> None: - assert __version__ == "0.17.0a0" + assert __version__ == "0.18.0a0" def test_package_includes_pep561_marker() -> None: diff --git a/tests/unit/tools/test_runtime_info.py b/tests/unit/tools/test_runtime_info.py index e3cbd53..aee4bf3 100644 --- a/tests/unit/tools/test_runtime_info.py +++ b/tests/unit/tools/test_runtime_info.py @@ -35,7 +35,7 @@ async def test_runtime_info_returns_safe_structured_data() -> None: payload = json.loads(result.content) assert result.tool_call_id == "call-1" assert result.is_error is False - assert payload["package_version"] == "0.17.0a0" + assert payload["package_version"] == "0.18.0a0" assert payload["python_version"] assert payload["platform"] diff --git a/tests/unit/web/__init__.py b/tests/unit/web/__init__.py new file mode 100644 index 0000000..8b13789 --- /dev/null +++ b/tests/unit/web/__init__.py @@ -0,0 +1 @@ + diff --git a/tests/unit/web/test_app.py b/tests/unit/web/test_app.py new file mode 100644 index 0000000..10a3001 --- /dev/null +++ b/tests/unit/web/test_app.py @@ -0,0 +1,201 @@ +from __future__ import annotations + +import asyncio +import json +from pathlib import Path + +import httpx +import pytest + +from mini_code_agent.agent.models import AgentResult, StopReason +from mini_code_agent.config import AppSettings +from mini_code_agent.providers.base import TokenUsage +from mini_code_agent.web.app import create_web_app +from mini_code_agent.web.manager import WebRunManager + + +def settings(tmp_path: Path) -> AppSettings: + return AppSettings.model_validate( + { + "data_dir": tmp_path / "data", + "provider": "openai_compatible", + "model": "Pro/zai-org/GLM-4.7", + "base_url": "https://api.siliconflow.cn/v1", + "openai_api_key": "server-only-secret", + } + ) + + +def result() -> AgentResult: + return AgentResult( + run_id="agent-run", + messages=(), + stop_reason=StopReason.COMPLETED, + turns=1, + tool_calls=0, + usage=TokenUsage(input_tokens=4, output_tokens=2), + final_text="Finished.", + ) + + +@pytest.fixture +def anyio_backend() -> str: + return "asyncio" + + +@pytest.mark.asyncio +async def test_health_and_bootstrap_expose_status_but_not_secret(tmp_path: Path) -> None: + async def runner(prompt: str, approval: object, events: object) -> AgentResult: + del prompt, approval, events + return result() + + app = create_web_app( + settings(tmp_path), + workspace=tmp_path, + manager=WebRunManager(runner), + csrf_token="fixed-token", + ) + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=app), + base_url="http://127.0.0.1:8765", + ) as client: + health = await client.get("/healthz") + bootstrap = await client.get("/api/bootstrap") + + assert health.json() == {"status": "ok"} + payload = bootstrap.json() + assert payload["workspace"] == str(tmp_path.resolve()) + assert payload["provider"] == "openai_compatible" + assert payload["model"] == "Pro/zai-org/GLM-4.7" + assert payload["api_key_configured"] is True + assert payload["csrf_token"] == "fixed-token" + assert "server-only-secret" not in bootstrap.text + + +@pytest.mark.asyncio +async def test_mutations_require_token_and_loopback_origin(tmp_path: Path) -> None: + async def runner(prompt: str, approval: object, events: object) -> AgentResult: + del prompt, approval, events + return result() + + app = create_web_app( + settings(tmp_path), + workspace=tmp_path, + manager=WebRunManager(runner), + csrf_token="fixed-token", + ) + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=app), + base_url="http://127.0.0.1:8765", + ) as client: + missing = await client.post("/api/runs", json={"prompt": "Inspect"}) + foreign = await client.post( + "/api/runs", + json={"prompt": "Inspect"}, + headers={ + "X-Mini-Code-Agent-Token": "fixed-token", + "Origin": "https://attacker.example", + }, + ) + accepted = await client.post( + "/api/runs", + json={"prompt": "Inspect"}, + headers={ + "X-Mini-Code-Agent-Token": "fixed-token", + "Origin": "http://localhost:8765", + }, + ) + + assert missing.status_code == 403 + assert foreign.status_code == 403 + assert accepted.status_code == 202 + + +@pytest.mark.asyncio +async def test_start_conflict_cancel_and_sse_replay(tmp_path: Path) -> None: + release = asyncio.Event() + + async def runner(prompt: str, approval: object, events: object) -> AgentResult: + del prompt, approval, events + await release.wait() + return result() + + manager = WebRunManager(runner) + app = create_web_app( + settings(tmp_path), + workspace=tmp_path, + manager=manager, + csrf_token="fixed-token", + ) + headers = {"X-Mini-Code-Agent-Token": "fixed-token"} + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=app), + base_url="http://127.0.0.1:8765", + ) as client: + started = await client.post( + "/api/runs", + json={"prompt": "Inspect"}, + headers=headers, + ) + run_id = started.json()["run_id"] + conflict = await client.post( + "/api/runs", + json={"prompt": "Second"}, + headers=headers, + ) + cancelled = await client.post( + f"/api/runs/{run_id}/cancel", + headers=headers, + ) + stream = await client.get(f"/api/runs/{run_id}/events?after=0") + + assert started.status_code == 202 + assert conflict.status_code == 409 + assert cancelled.status_code == 200 + assert stream.headers["content-type"].startswith("text/event-stream") + data_lines = [ + line.removeprefix("data: ") + for line in stream.text.splitlines() + if line.startswith("data: ") + ] + events = [json.loads(line) for line in data_lines] + assert [event["type"] for event in events] == [ + "web_run_started", + "web_run_cancelled", + ] + + +@pytest.mark.asyncio +async def test_approval_route_returns_not_found_for_stale_decision( + tmp_path: Path, +) -> None: + async def runner(prompt: str, approval: object, events: object) -> AgentResult: + del prompt, approval, events + return result() + + manager = WebRunManager(runner) + app = create_web_app( + settings(tmp_path), + workspace=tmp_path, + manager=manager, + csrf_token="fixed-token", + ) + headers = {"X-Mini-Code-Agent-Token": "fixed-token"} + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=app), + base_url="http://127.0.0.1:8765", + ) as client: + started = await client.post( + "/api/runs", + json={"prompt": "Inspect"}, + headers=headers, + ) + run_id = started.json()["run_id"] + await manager.wait(run_id) + response = await client.post( + f"/api/runs/{run_id}/approvals/stale", + json={"approved": True}, + headers=headers, + ) + + assert response.status_code == 409 diff --git a/tests/unit/web/test_manager.py b/tests/unit/web/test_manager.py new file mode 100644 index 0000000..d8273fe --- /dev/null +++ b/tests/unit/web/test_manager.py @@ -0,0 +1,165 @@ +from __future__ import annotations + +import asyncio +import json + +import pytest + +from mini_code_agent.agent.events import RunStarted +from mini_code_agent.agent.models import AgentResult, StopReason +from mini_code_agent.policy.models import ( + ActionPreview, + ApprovalRequest, + RiskLevel, +) +from mini_code_agent.providers.base import TokenUsage +from mini_code_agent.tools.base import SideEffect +from mini_code_agent.web.manager import ( + RunConflictError, + WebRunManager, + WebRunStatus, +) + + +def completed_result(*, final_text: str = "done") -> AgentResult: + return AgentResult( + run_id="agent-run", + messages=(), + stop_reason=StopReason.COMPLETED, + turns=1, + tool_calls=0, + usage=TokenUsage(input_tokens=4, output_tokens=2), + final_text=final_text, + ) + + +@pytest.mark.asyncio +async def test_manager_publishes_monotonic_redacted_lifecycle() -> None: + secret_prompt = "inspect project with sk-live-secret" + + async def runner(prompt: str, approval: object, events: object) -> AgentResult: + del approval + assert prompt == secret_prompt + events.publish(RunStarted(run_id="agent-run", max_turns=8)) # type: ignore[attr-defined] + return completed_result() + + manager = WebRunManager(runner) + snapshot = await manager.start(secret_prompt) + terminal = await manager.wait(snapshot.run_id) + recorded = manager.events_after(snapshot.run_id) + + assert terminal.status is WebRunStatus.COMPLETED + assert [event.sequence for event in recorded] == list(range(1, len(recorded) + 1)) + assert [event.type for event in recorded] == [ + "web_run_started", + "agent_event", + "web_run_completed", + ] + serialized = json.dumps( + [event.model_dump(mode="json") for event in recorded], + ensure_ascii=False, + ) + assert secret_prompt not in serialized + assert "sk-live-secret" not in serialized + assert recorded[-1].payload["final_text"] == "done" + + +@pytest.mark.asyncio +async def test_manager_allows_only_one_active_run() -> None: + release = asyncio.Event() + + async def runner(prompt: str, approval: object, events: object) -> AgentResult: + del prompt, approval, events + await release.wait() + return completed_result() + + manager = WebRunManager(runner) + first = await manager.start("first") + + with pytest.raises(RunConflictError): + await manager.start("second") + + release.set() + await manager.wait(first.run_id) + second = await manager.start("second") + release.set() + await manager.wait(second.run_id) + + +@pytest.mark.asyncio +async def test_approval_is_bounded_single_use_and_can_be_approved() -> None: + request = ApprovalRequest( + preview=ActionPreview( + tool_call_id="call-1", + tool_name="run_command", + side_effect=SideEffect.EXECUTE, + risk=RiskLevel.HIGH, + summary="Run focused tests", + reason="Verify the implementation", + resources=("tests/unit/web/test_manager.py",), + command=("python", "-m", "pytest"), + diff="+ added\n- removed", + ), + rule_id="cli-ask-execute", + rationale="Commands require explicit approval.", + ) + approval_started = asyncio.Event() + + async def runner(prompt: str, approval: object, events: object) -> AgentResult: + del prompt, events + approval_started.set() + approved = await approval.approve(request) # type: ignore[attr-defined] + return completed_result(final_text=f"approved={approved}") + + manager = WebRunManager(runner) + snapshot = await manager.start("test") + await approval_started.wait() + await asyncio.sleep(0) + + approval_event = manager.events_after(snapshot.run_id)[-1] + assert approval_event.type == "approval_required" + assert approval_event.payload["preview"]["tool_call_id"] == "call-1" + assert len(approval_event.payload["preview"]["diff"]) <= 32_768 + assert await manager.decide_approval(snapshot.run_id, "missing", True) is False + assert await manager.decide_approval(snapshot.run_id, "call-1", True) is True + assert await manager.decide_approval(snapshot.run_id, "call-1", False) is False + + terminal = await manager.wait(snapshot.run_id) + assert terminal.status is WebRunStatus.COMPLETED + assert manager.events_after(snapshot.run_id)[-1].payload["final_text"] == ("approved=True") + + +@pytest.mark.asyncio +async def test_cancel_rejects_pending_approval_and_emits_one_terminal_event() -> None: + approval_waiting = asyncio.Event() + + async def runner(prompt: str, approval: object, events: object) -> AgentResult: + del prompt, events + approval_waiting.set() + await approval.approve( # type: ignore[attr-defined] + ApprovalRequest( + preview=ActionPreview( + tool_call_id="call-cancel", + tool_name="write_file", + side_effect=SideEffect.WRITE, + risk=RiskLevel.MEDIUM, + summary="Write a file", + ), + rule_id="write-ask", + rationale="Writes require approval.", + ) + ) + return completed_result() + + manager = WebRunManager(runner) + snapshot = await manager.start("cancel me") + await approval_waiting.wait() + await asyncio.sleep(0) + + assert await manager.cancel(snapshot.run_id) is True + terminal = await manager.wait(snapshot.run_id) + events = manager.events_after(snapshot.run_id) + + assert terminal.status is WebRunStatus.CANCELLED + assert [event.type for event in events].count("web_run_cancelled") == 1 + assert await manager.decide_approval(snapshot.run_id, "call-cancel", approved=True) is False diff --git a/tests/unit/web/test_models.py b/tests/unit/web/test_models.py new file mode 100644 index 0000000..d91899c --- /dev/null +++ b/tests/unit/web/test_models.py @@ -0,0 +1,25 @@ +import pytest +from pydantic import ValidationError + +from mini_code_agent.web.models import ( + ApprovalDecisionRequest, + StartRunRequest, + WebEvent, +) + + +def test_start_run_request_rejects_blank_and_oversized_prompts() -> None: + with pytest.raises(ValidationError): + StartRunRequest(prompt="") + + with pytest.raises(ValidationError): + StartRunRequest(prompt="x" * 20_001) + + +def test_web_models_are_frozen_and_bounded() -> None: + event = WebEvent(sequence=1, type="web_run_started", payload={"status": "running"}) + + with pytest.raises(ValidationError): + event.sequence = 2 + + assert ApprovalDecisionRequest(approved=True).approved is True diff --git a/tests/unit/web/test_static.py b/tests/unit/web/test_static.py new file mode 100644 index 0000000..b582bdb --- /dev/null +++ b/tests/unit/web/test_static.py @@ -0,0 +1,53 @@ +from importlib import resources + + +def static_text(name: str) -> str: + return ( + resources.files("mini_code_agent.web").joinpath("static", name).read_text(encoding="utf-8") + ) + + +def test_frontend_resources_are_packaged_without_remote_dependencies() -> None: + html = static_text("index.html") + css = static_text("styles.css") + javascript = static_text("app.js") + combined = "\n".join((html, css, javascript)) + + assert "https://" not in combined + assert "http://" not in combined + assert "cdn" not in combined.lower() + + +def test_frontend_contains_workbench_landmarks_and_controls() -> None: + html = static_text("index.html") + + for marker in ( + 'id="session-rail"', + 'id="transcript"', + 'id="prompt-input"', + 'id="run-button"', + 'id="cancel-button"', + 'id="inspector"', + 'id="activity-list"', + 'id="approval-panel"', + 'id="changes-panel"', + ): + assert marker in html + assert 'type="password"' not in html + assert "api key" not in html.lower() + + +def test_frontend_renders_untrusted_values_without_dynamic_html() -> None: + javascript = static_text("app.js") + + assert ".textContent" in javascript + assert "innerHTML" not in javascript + assert "insertAdjacentHTML" not in javascript + + +def test_styles_define_desktop_and_mobile_workbench_tracks() -> None: + css = static_text("styles.css") + + assert "--topbar-height" in css + assert "grid-template-columns" in css + assert "@media (max-width: 720px)" in css diff --git a/uv.lock b/uv.lock index 5af5b56..667f2d1 100644 --- a/uv.lock +++ b/uv.lock @@ -206,6 +206,22 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/07/6c/aa3f2f849e01cb6a001cd8554a88d4c77c5c1a31c95bdf1cf9301e6d9ef4/defusedxml-0.7.1-py2.py3-none-any.whl", hash = "sha256:a352e7e428770286cc899e2542b6cdaedb2b4953ff269a210103ec58f6198a61", size = 25604, upload-time = "2021-03-08T10:59:24.45Z" }, ] +[[package]] +name = "fastapi" +version = "0.139.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "annotated-doc" }, + { name = "pydantic" }, + { name = "starlette" }, + { name = "typing-extensions" }, + { name = "typing-inspection" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/d3/af/a5f50ccfa659ec1802cb4ca842c23f06d906a8cc9aef6016a2caeea3d4ed/fastapi-0.139.0.tar.gz", hash = "sha256:99ab7b2d92223c76d6cf10757ab3f89d45b38267fc20b2a136cf02f6beac3145", size = 423016, upload-time = "2026-07-01T16:35:33.436Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/9e/7c/8e3c6ad324ea5cb36604fc3f968554887891c316d9dfde57761611d907ad/fastapi-0.139.0-py3-none-any.whl", hash = "sha256:cf15e1e9e667ddb0ad63811e60bd11390d1aac838ca4a7a23f421807b2308189", size = 130339, upload-time = "2026-07-01T16:35:32.19Z" }, +] + [[package]] name = "h11" version = "0.16.0" @@ -360,10 +376,11 @@ wheels = [ [[package]] name = "mini-code-agent" -version = "0.17.0a0" +version = "0.18.0a0" source = { editable = "." } dependencies = [ { name = "defusedxml" }, + { name = "fastapi" }, { name = "httpx" }, { name = "httpx-sse" }, { name = "jsonschema" }, @@ -374,6 +391,7 @@ dependencies = [ { name = "pyyaml" }, { name = "rich" }, { name = "typer" }, + { name = "uvicorn" }, ] [package.dev-dependencies] @@ -392,6 +410,7 @@ dev = [ [package.metadata] requires-dist = [ { name = "defusedxml", specifier = ">=0.7.1,<0.8" }, + { name = "fastapi", specifier = ">=0.116,<1" }, { name = "httpx", specifier = ">=0.28,<1" }, { name = "httpx-sse", specifier = ">=0.4,<1" }, { name = "jsonschema", specifier = ">=4.23,<5" }, @@ -402,6 +421,7 @@ requires-dist = [ { name = "pyyaml", specifier = ">=6.0.2,<7" }, { name = "rich", specifier = ">=13.9,<15" }, { name = "typer", specifier = ">=0.15,<1" }, + { name = "uvicorn", specifier = ">=0.34,<1" }, ] [package.metadata.requires-dev]