diff --git a/CHANGELOG.md b/CHANGELOG.md index f18cc54..afc9632 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,58 @@ All notable changes follow Keep a Changelog. Versions follow Semantic Versioning ## [Unreleased] +## [0.15.0-alpha.0] - 2026-07-02 + +### Added + +- Immutable host-owned `SubagentProfile` contracts with exact child Tools, independent Agent/ + batch limits, no recursion, and `TrustSource.SUBAGENT`. +- `SubagentSupervisor` with fresh child contexts, preflight composition, structured + `asyncio.TaskGroup` concurrency, per-child and outer deadlines, ordered aggregation, and + cancellation propagation. +- Bounded child/batch result models, canonical SHA-256 projections, ToolResult evidence hashes, + static failures, and metadata-only lifecycle events. +- One dynamic governed parent analysis Tool per profile, with profile-specific JSON Schema, + medium-risk preview, canonical ASCII result JSON, and UTF-8 byte budget. +- Real parent/child Agent integration using governed `ReadFileTool`/`SearchTextTool`, including + Policy deny, non-recursion, sibling timeout isolation, event omission, byte-identical Workspace, + and parent cancellation. +- M6a architecture, threat-model, ADR, learning, and resume documentation. + +### Changed + +- Added `TrustSource.SUBAGENT` so child Tool Policy is independently addressable from parent model + and extension calls. +- The stable installed-package smoke imports the public Subagent profile, supervisor, Tool, and + builder API. +- Subagent unit tests use a package namespace so Pytest's default import mode can collect the full + suite alongside existing same-named test modules. + +### Security + +- Every child ID, Provider, and governed Tool executor is validated before child Provider I/O; + malformed/duplicate IDs, reused objects, capability drift, non-read-only definitions, and + non-SUBAGENT provenance fail the complete batch. +- Duplicate, empty, NUL-containing, oversized, or excessive tasks fail before child composition. +- Child sessions are non-interactive, cannot receive a delegation Tool, and cannot convert `ASK` + into authority. +- Child/batch timeout is isolated, external cancellation is re-raised, and no detached asyncio + task survives the parent ToolCall. +- Events exclude task/prompt/message/summary/argument/result/exception content; evidence retains + only bounded metadata and SHA-256. +- In-process children are not an OS sandbox. M6a does not claim Tool-implementation isolation, + durable parent-child audit, semantic proof from hashes, rollback, or exactly-once behavior. + +### Verification + +- Python 3.12.13 passed 1060 tests with 10 Windows symlink-privilege skips and 91.08% branch + coverage before the final two hardening regressions; the complete focused Subagent/integration + suite then passed 100 tests. +- Final Python 3.13.14 passed 1062 tests with the same 10 platform skips and 91.09% branch + coverage. Ruff format/check, strict Pyright, Bandit, and locked runtime pip-audit passed. +- Remote CI, reproducible artifact, installed wheel/sdist, tag, and GitHub prerelease evidence is + pending the release task and is not claimed here. + ## [0.14.0-alpha.0] - 2026-07-01 ### Added diff --git a/README.md b/README.md index 79fbfbd..93a4d05 100644 --- a/README.md +++ b/README.md @@ -2,15 +2,15 @@ A framework-light, provider-neutral coding agent built from first principles. -> Status: pre-alpha. M5b provides a provider-neutral Agent Core, Anthropic/OpenAI-compatible +> Status: pre-alpha. M6a provides a provider-neutral Agent Core, Anthropic/OpenAI-compatible > adapters, a schema-validating Tool Registry, a cross-platform Workspace boundary, bounded > Read/Search, conflict-aware Write/Edit, policy-governed argv command execution, and deterministic > context admission, hardened read-only Git evidence, governed Pytest diagnostics, versioned SQLite > Session/Trace persistence, fail-closed Checkpoint/Resume, and a host-controlled bounded Repair > loop, provenance-aware lazy Skills, deterministic host-registered Tool Hooks, and host-pinned -> local MCP stdio Tools. OS sandboxing, shell-string execution, project-provided executable Hooks, -> automatic Repair resume, remote HTTP/OAuth MCP, Subagents/Worktrees, and live-provider CI are not -> implemented. +> local MCP stdio Tools, and bounded host-profiled read-only analysis Subagents. OS sandboxing, +> shell-string execution, project-provided executable Hooks, automatic Repair resume, remote +> HTTP/OAuth MCP, write-capable Subagents/Worktrees, and live-provider CI are not implemented. ## Requirements @@ -236,6 +236,27 @@ approval are not OS sandboxing. Remote HTTP/OAuth, Resources, Prompts, Roots, Sa Elicitation, Tasks, dynamic Tool lists, and package installation are not supported. See `docs/architecture/governed-mcp.md`. +## Governed Analysis Subagents + +M6a exposes one governed parent Tool per immutable host profile. The model supplies one to four +unique bounded tasks; the host fixes the child system prompt, exact read-only Tool names, Agent +limits, concurrency, deadlines, and result budgets. + +Before any child Provider request, `SubagentSupervisor` requires distinct Providers/executors, +exact `READ_ONLY` definitions, `governance_enforced is True`, and +`TrustSource.SUBAGENT` for every child Tool. Each child receives one fresh task message, not the +parent or sibling transcript. Delegation Tools are structurally unavailable to children. + +All children belong to one `asyncio.TaskGroup`. A semaphore bounds concurrency, individual and +batch timeouts have typed outcomes, input order is preserved, and external cancellation cancels +and joins every child before being re-raised. Parent results contain bounded untrusted summaries +and ToolResult metadata/SHA-256 evidence, not raw child transcripts or Tool content. Events omit +tasks, prompts, summaries, arguments, results, repository content, and exception text. + +In-process context isolation is not an OS sandbox. M6a cannot write, run commands, call network +Tools, open nested approval prompts, persist durable child traces, create Worktrees, or merge +changes. See `docs/architecture/governed-subagents.md`. + ## Documentation - Product design: `docs/superpowers/specs/2026-06-29-mini-code-agent-design.md` @@ -255,6 +276,7 @@ Elicitation, Tasks, dynamic Tool lists, and package installation are not support - Bounded Repair loop: `docs/architecture/bounded-repair-loop.md` - Governed Skills and Hooks: `docs/architecture/governed-extensions.md` - Governed MCP stdio: `docs/architecture/governed-mcp.md` +- Governed analysis Subagents: `docs/architecture/governed-subagents.md` - Threat model: `docs/architecture/threat-model.md` - Provider protocol ADR: `docs/adr/0002-provider-wire-protocols.md` - Workspace boundary ADR: `docs/adr/0003-workspace-boundary.md` @@ -268,6 +290,7 @@ Elicitation, Tasks, dynamic Tool lists, and package installation are not support - Host-controlled bounded Repair ADR: `docs/adr/0011-host-controlled-bounded-repair.md` - Inert Skills and host Hooks ADR: `docs/adr/0012-inert-skills-host-hooks.md` - Host-pinned stdio MCP ADR: `docs/adr/0013-host-pinned-stdio-mcp.md` +- Bounded host-profiled Subagents ADR: `docs/adr/0014-bounded-host-profiled-subagents.md` ## License diff --git a/SECURITY.md b/SECURITY.md index 39ba9bd..565734e 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -11,9 +11,9 @@ after the repository is published. Until then, contact the repository owner priv ## Current Boundary -Model output, repository content, project Skills, Tool arguments, test reports, and MCP servers -are untrusted inputs. File, command, Git, test, Repair, and MCP Tool actions pass typed validation, -Policy, and approval where applicable. +Model output, repository content, project Skills, Tool arguments, test reports, MCP servers, child +Agent output, and child summaries are untrusted inputs. File, command, Git, test, Repair, MCP, and +Subagent Tool actions pass typed validation, Policy, and approval where applicable. M5a Skills are inert Markdown data. Discovery rejects links/reparse points, unsafe YAML, invalid metadata, conflicts, drift, and resource-limit violations. Parsing or hashing a Skill does not @@ -44,6 +44,24 @@ Timeout/cancellation cannot prove that a remote side effect did not complete. Th support remote HTTP/OAuth MCP, package installation, executable signatures, dynamic Tool lists, Resources, Prompts, Roots, Sampling, Elicitation, or Tasks. +M6a analysis Subagents use immutable host profiles, fresh child contexts, exact read-only Tool +sets, independent Agent/result budgets, and `TrustSource.SUBAGENT`. Every child Tool executor must +prove governance and SUBAGENT provenance before any child Provider request. Child sessions are +non-interactive, so `ASK` fails closed instead of opening nested approval. Parent Policy deny +prevents child factories and Provider I/O. + +All child tasks belong to one `asyncio.TaskGroup`; child and batch deadlines are separate, and +external cancellation is re-raised after children are cancelled and joined. Results expose only +bounded untrusted summaries and ToolResult metadata/SHA-256 evidence. Subagent events exclude +tasks, prompts, messages, summaries, Tool arguments/results, repository content, and exception +text. + +M6a children are in-process and are not an OS, process, memory, credential, or network sandbox. +Read-only Tool admission does not constrain malicious host-supplied Provider or Tool code. +Evidence hashes are not signatures, semantic validation, confidentiality, or durable audit. +M6a does not support child writes, command/network Tools, recursive delegation, Worktrees, +candidate adoption, merge, rollback, or exactly-once execution. + The project does not claim OS-level sandboxing unless an explicit sandbox backend is enabled and documented. It also does not claim that Hook timeout stops work delegated to another thread or process, or that SHA-256 establishes extension authorship. diff --git a/docs/adr/0014-bounded-host-profiled-subagents.md b/docs/adr/0014-bounded-host-profiled-subagents.md new file mode 100644 index 0000000..d1a836f --- /dev/null +++ b/docs/adr/0014-bounded-host-profiled-subagents.md @@ -0,0 +1,109 @@ +# ADR 0014: Use Bounded Host-Profiled Analysis Subagents + +- Status: Accepted +- Date: 2026-07-02 + +## Context + +Independent code-reading tasks can be parallelized, and a child Agent can keep exploratory +messages out of the parent transcript. A naive Subagent feature, however, can duplicate parent +authority, inherit unrelated context, recursively create more Agents, start background approval +prompts, swallow cancellation, leak raw Tool content into aggregation, or leave orphan tasks. + +The project already has a typed `AgentRuntime`, Tool Registry, Workspace boundary, Policy, +provenance, and deterministic limits. A Subagent design should reuse these contracts rather than +introduce a second execution and authorization system. + +Write-capable children introduce a different problem: concurrent mutation, repository identity, +candidate persistence, merge/adoption authority, and cleanup uncertainty. Combining that problem +with initial analysis delegation would make the first boundary too broad. + +## Decision + +M6a implements host-profiled, in-process, non-recursive analysis children. + +The trusted host creates one immutable `SubagentProfile` per parent Tool. It fixes the child +system prompt, exact ordered Tool names, Agent limits, concurrency, deadlines, evidence, summary, +and result budgets. The model supplies only one to four unique bounded tasks plus a reason. + +Before any child Provider request, `SubagentSupervisor` creates and validates every child: + +- unique host child ID; +- distinct Provider and governed Tool executor; +- exact `READ_ONLY` definitions; +- `governance_enforced is True`; +- `TrustSource.SUBAGENT` for every child Tool; +- no delegation Tool. + +Every child gets a fresh one-message context and an independent `AgentRuntime`. All child tasks +belong to one `asyncio.TaskGroup`; a semaphore bounds concurrency, child and batch timeouts are +separate, results are stored by input ordinal, and external `CancelledError` is re-raised. + +Children run in `SessionMode.NON_INTERACTIVE`, so a child Policy `ASK` cannot open a nested prompt +and fails closed. + +The parent receives only bounded typed projections: untrusted final summaries, usage/counts, +static failure metadata, and SHA-256 evidence for correlated ToolResult content. Lifecycle events +contain metadata and hashes but never task text, prompts, messages, summaries, Tool arguments, +ToolResult content, repository content, or exception text. + +M6a is read-only. Worktree-backed implementation children and candidate adoption are deferred to +M6b. + +## Consequences + +Positive: + +- child authority is an exact host capability profile, not copied parent authority; +- parent and sibling transcripts are not implicitly shared; +- Policy can distinguish parent-model calls from delegated calls; +- recursive delegation is structurally unavailable; +- one child timeout/failure does not erase sibling results; +- TaskGroup gives one lexical owner for cancellation and joining; +- input order remains stable under out-of-order completion; +- aggregation does not copy raw repository or ToolResult content; +- parent Policy deny prevents child factories and Provider I/O; +- the existing Agent/Tool/Workspace contracts remain the execution path. + +Negative: + +- each task adds a Provider session and may increase cost, latency, and rate-limit pressure; +- fresh children repeat context that a forked child might otherwise inherit; +- in-process Provider and Tool implementations share memory and OS authority; +- read-only admission cannot sandbox malicious host code; +- evidence hashes do not validate semantic correctness; +- child events are best-effort and lack durable parent run/turn linkage; +- `NON_INTERACTIVE` means child work requiring approval cannot proceed; +- M6a cannot implement or merge code changes. + +## Alternatives Rejected + +- **Give children the parent transcript:** leaks unrelated context, weakens attribution, and makes + context growth implicit. +- **Let the model define child prompts and Tools:** allows model output to create authority. +- **Copy the complete parent Tool Registry:** can expose writes, commands, network access, MCP, or + delegation without a separate decision. +- **Recursive delegation with a depth counter:** a depth limit does not solve capability + amplification, cost fan-out, or audit complexity; M6a admits no delegation Tool. +- **Detached `asyncio.create_task`:** permits orphan work and ambiguous cancellation ownership. +- **`asyncio.gather(return_exceptions=True)`:** makes cancellation and task-lifetime invariants + less explicit than TaskGroup plus typed child outcomes. +- **One timeout only:** cannot distinguish a slow child from a whole-batch deadline. +- **Interactive child approval:** background children must not compete for user prompts or reuse + parent approval. +- **Return complete child transcripts:** consumes parent context and leaks Tool arguments/results + rather than a bounded projection. +- **Run every child in a subprocess now:** stronger interpreter isolation also requires + credential transport, authenticated IPC, Provider lifecycle, process cleanup, and durable + result protocols; deferred until the in-process contract is stable. +- **Enable writes in the parent checkout:** concurrent children can collide with user work and + one another. M6b requires host-managed Worktrees and explicit candidate adoption. +- **Adopt a multi-agent framework:** would duplicate or obscure the project's existing + Provider/Tool/Policy semantics and cancellation evidence. + +## Follow-up + +M6b may add one write-capable implementation profile only through host-created, locked, +no-checkout Git Worktrees. A child result will produce a bounded candidate snapshot; a separate +parent-side approval will control adoption. No M6b change may weaken M6a's exact profiles, +SUBAGENT provenance, cancellation propagation, or no-recursion rule. diff --git a/docs/architecture/governed-subagents.md b/docs/architecture/governed-subagents.md new file mode 100644 index 0000000..f6492a5 --- /dev/null +++ b/docs/architecture/governed-subagents.md @@ -0,0 +1,286 @@ +# Governed Analysis Subagents + +## Purpose and Scope + +M6a lets one parent Agent delegate one to four independent analysis tasks without giving child +Agents new authority. It adds bounded concurrency and context isolation while preserving the +existing Tool governance boundary. + +M6a supports only: + +- host-created immutable analysis profiles; +- fresh in-process child `AgentRuntime` instances; +- exact read-only child Tool sets; +- `TrustSource.SUBAGENT` provenance; +- one governed parent Tool per profile; +- structured fan-out/fan-in with per-child and outer batch deadlines; +- bounded summaries, hashed Tool-result evidence, and metadata-only lifecycle events. + +M6a does not create Git Worktrees and does not permit child writes, command execution, network +Tools, nested delegation, background approval prompts, durable child checkpoints, or autonomous +merging. Those are separate designs, not hidden capabilities. + +## Authority and Data Flow + +The trusted host owns the profile, factories, workspace root, Provider selection, Tool +composition, limits, and event sink. The parent model supplies only task text and a bounded +reason. Repository content, parent/child model output, Tool arguments, Tool results, and child +summaries remain untrusted. + +```mermaid +sequenceDiagram + participant Parent as "Parent Agent" + participant Policy as "Parent Tool governance" + participant Tool as "SubagentAnalysisTool" + participant Supervisor as "SubagentSupervisor" + participant Child as "Fresh child AgentRuntime" + participant ChildPolicy as "SUBAGENT Tool governance" + participant Workspace as "Read-only Workspace" + + Parent->>Policy: "delegate_analysis(tasks, reason)" + Policy->>Policy: "Schema, preview, Hook, Policy" + Policy->>Tool: "allowed ToolCall" + Tool->>Supervisor: "run_batch(parent call ID, tasks)" + Supervisor->>Supervisor: "validate tasks and compose all children" + par "bounded TaskGroup" + Supervisor->>Child: "fresh user message + host system prompt" + Child->>ChildPolicy: "exact read/search ToolCall" + ChildPolicy->>ChildPolicy: "NON_INTERACTIVE + SUBAGENT Policy" + ChildPolicy->>Workspace: "bounded read-only operation" + Workspace-->>Child: "ToolResult" + Child-->>Supervisor: "AgentResult" + end + Supervisor->>Supervisor: "ordered projection + evidence hashes" + Supervisor-->>Tool: "SubagentBatchResult" + Tool-->>Parent: "bounded canonical JSON" +``` + +The parent Tool is declared `READ_ONLY` because admitted child capabilities cannot mutate the +Workspace or invoke execute/network Tools. Its preview is `MEDIUM` risk and exposes only the +profile, task count, reason, and resource `"."`; it does not copy the task array into the +preview. + +The parent Tool normally enters `GovernedToolExecutor` with `TrustSource.MODEL`. Child Tools use a +separate executor whose default provenance is `TrustSource.SUBAGENT`. A Policy can therefore +allow ordinary model reads while applying a different rule to delegated reads. + +## Host Profile + +`SubagentProfile` is immutable trusted composition data. It contains: + +- `profile_id`: stable host identity; +- `local_name`: the parent Agent Tool name; +- `description`: host-authored model-facing description; +- `system_prompt`: child-only host instructions; +- `tool_names`: exact ordered child Tool names; +- `AgentLimits`: child turns, ToolCalls, and Provider/Tool deadlines; +- `SubagentLimits`: batch size, concurrency, task/result/evidence budgets, and deadlines. + +Profiles reject duplicate Tool names, their own local delegation name, every child name beginning +with `delegate_`, Agent limits above the M6a ceiling, and configurations that cannot retain one +evidence item per possible ToolCall. + +Hard ceilings are: + +| Resource | Ceiling | +|---|---:| +| Tasks per batch | 4 | +| Concurrent children | 4 | +| Task characters | 20,000 | +| Child turns | 32 | +| Child ToolCalls | 128 | +| Child timeout | 600 seconds | +| Batch timeout | 900 seconds | +| Summary characters | 32,000 | +| Evidence items | 256 | +| Parent ToolResult | 1 MiB | + +The batch deadline cannot be lower than the child deadline. A host can select lower values for a +specific profile. + +## Composition Before Execution + +`SubagentSupervisor` receives two host factories: + +```python +class SubagentProviderFactory(Protocol): + def create( + self, + profile: SubagentProfile, + child_id: str, + ) -> ModelProvider: ... + + +class SubagentToolFactory(Protocol): + def create( + self, + profile: SubagentProfile, + workspace_root: Path, + ) -> ToolExecutor: ... +``` + +Before any child Provider request, the supervisor: + +1. validates the parent ToolCall ID and the complete task tuple; +2. rejects empty, duplicate, NUL-containing, excessive, or oversized tasks; +3. generates the complete child-ID tuple and rejects malformed or duplicate IDs; +4. creates a distinct Provider and Tool executor for every child; +5. rejects reused Provider or executor object identity; +6. requires exact ordered Tool names; +7. requires every child definition to be `READ_ONLY`; +8. requires `governance_enforced is True`; +9. requires `trust_source_for(name) is TrustSource.SUBAGENT`; +10. constructs one independent `AgentRuntime` per child. + +Any mismatch raises one static `SubagentCompositionError`. Raw factory exceptions and capability +details are not returned to the parent model. + +`build_subagent_tools()` also rejects duplicate profile IDs, duplicate parent local names, and any +parent local name that appears in any child Tool set. This prevents a child capability graph from +containing a route back into delegation even when multiple profiles are composed together. + +## Fresh Context, Not Forked Context + +Each child starts with exactly: + +- the host profile's `system_prompt`; +- one fresh `Message.user_text(task)`; +- the exact child Tool definitions. + +It does not receive the parent transcript, parent system prompt, sibling tasks, sibling messages, +parent Tool results, or another child's context. This reduces accidental context sharing and +makes child evidence attributable to one task. + +Fresh context is isolation of Agent messages, not confidentiality from the Provider or operating +system. A child can still read every file its admitted Tools allow, and an in-process Provider or +Tool implementation has the Python process's authority. + +## Structured Concurrency + +One outer `asyncio.TaskGroup` owns every child task. A semaphore limits active child execution to +`max_concurrency`, and one ordinal result slot preserves input order even when completion order +differs. + +Each child uses `asyncio.timeout(child_timeout_seconds)`. Ordinary timeout or projection failure +becomes that child's typed result so siblings continue. One outer +`asyncio.timeout(batch_timeout_seconds)` cancels unfinished TaskGroup members and fills every +unfinished ordinal with `BATCH_TIMED_OUT`. + +External cancellation is different from a deadline: + +- `CancelledError` is explicitly re-raised by the child runner and parent Tool; +- TaskGroup cancels and joins all children before control returns; +- no detached `create_task`, daemon thread, process, or orphan Agent survives the parent ToolCall. + +The implementation does not use `asyncio.gather(..., return_exceptions=True)` because that makes +task ownership and cancellation failure modes easier to obscure. + +## Results and Evidence + +`SubagentChildResult` contains bounded typed metadata: + +- child ID, ordinal, profile, status, and stop reason; +- turns, ToolCall count, and token usage; +- `untrusted_summary`; +- zero or more `SubagentEvidenceItem` records; +- a static error code/message for timeout or failure; +- a canonical result SHA-256. + +The summary is the child's final model text, truncated to the profile character budget and +explicitly labelled untrusted. It can contain mistakes or prompt injection and must not be treated +as authorization or test evidence. + +For every completed ToolCall, evidence retains only: + +- ToolCall ID and Tool name; +- error flag; +- result character count; +- SHA-256 of the UTF-8 ToolResult content. + +Evidence extraction validates call/result correlation, uniqueness, completeness, and count +against the Agent result. It never retains Tool arguments or raw ToolResult content. A hash proves +that two observed byte strings are equal; it does not prove that the content is true, safe, +authentic, or confidential. + +`SubagentBatchResult` preserves child ordinal order, validates profile/ID/count consistency, and +hashes a canonical ASCII JSON projection. `SubagentAnalysisTool` revalidates the returned model, +adds `content_type: "subagent_batch_result"`, serializes sorted compact ASCII JSON with +non-finite numbers disabled, and enforces the UTF-8 byte budget before returning to the parent. + +No token-saving percentage is claimed. A fresh child can reduce parent transcript pressure for +some workloads, but it also adds Provider calls, Tool calls, latency, and result summaries. +Benchmarking requires a fixed task corpus and Provider-specific token accounting. + +## Metadata-Only Events + +The supervisor emits best-effort: + +- `SubagentBatchStarted`; +- `SubagentStarted`; +- `SubagentCompleted`; +- `SubagentBatchCompleted`. + +Events include bounded IDs, profile, ordinal, status, duration, counts, usage, and result hashes. +They exclude task text, reason, system prompt, summary, messages, Tool arguments, ToolResult +content, repository text, and exception text. Sink exceptions do not change execution. + +These events are not added to the existing durable Agent journal because the parent +`ToolExecutor` contract does not carry parent run/turn context. M6a therefore does not claim +durable parent-child trace linkage. + +## Failure Matrix + +| Boundary | Failure | Public outcome | +|---|---|---| +| Parent Registry | unknown Tool or invalid JSON arguments | static Tool error; no supervisor call | +| Parent Policy | deny or non-interactive ask | `permission_denied`; no child composition | +| Batch validation | empty, duplicate, excessive, oversized, or NUL task | `invalid_batch` | +| Child IDs | malformed or duplicate host ID | static composition failure; no Provider request | +| Provider/Tool factory | exception, reused object, invalid protocol | static composition failure | +| Child capability | missing/extra/reordered/non-read-only/non-SUBAGENT Tool | static composition failure | +| Child run | non-completed stop reason | ordered `stopped` child result | +| Child deadline | timeout | ordered `timed_out` child result; siblings continue | +| Child projection | malformed result/evidence/summary | ordered `failed` child result | +| Batch deadline | unfinished children | ordered `batch_timed_out` results | +| Parent cancellation | caller cancellation | cancel/join children and re-raise | +| Result serialization | malformed or above byte budget | static failed/too-large Tool error | +| Event sink | ordinary exception | ignored; result is unchanged | + +## Operational Verification + +The real integration test uses: + +- a parent `AgentRuntime`; +- a governed `delegate_analysis` Tool; +- two child `AgentRuntime` instances; +- real governed `ReadFileTool` and `SearchTextTool`; +- one shared read-only Workspace; +- scripted, credential-free Providers. + +It proves fresh child contexts, distinct run IDs, exact child definitions, SUBAGENT provenance, +real read/search evidence hashes, ordered parent JSON, event omission, and byte-identical +Workspace state. Additional cases prove parent Policy deny causes zero factory calls, a child +cannot call the parent delegation Tool, one child timeout does not stop its sibling, and parent +cancellation cancels both children. + +## Threat Boundary and Non-Claims + +- In-process children are not an OS, process, memory, credential, or network sandbox. +- Read-only Tool admission constrains calls through the executor; it cannot constrain malicious + host-supplied Provider or Tool implementation code. +- `NON_INTERACTIVE` prevents nested approval prompts. It does not turn denied child work into + approved work; `ASK` fails closed. +- Context isolation does not hide repository data from a child Provider after a Tool returns it. +- Timeouts stop waiting and cancel cooperative work; they do not terminate arbitrary threads or + prove that an external Provider request had no cost. +- Evidence and result hashes are integrity fingerprints, not signatures, encryption, provenance, + semantic validation, or durable audit. +- Best-effort events do not establish crash recovery or exactly-once execution. +- M6a does not provide Worktree isolation, child writes, stage/commit/merge, candidate adoption, + conflict resolution, or rollback. +- M6a does not claim lower cost, lower latency, higher answer quality, or token savings without a + reproducible benchmark. + +M6b will treat write-capable implementation children as a separate authority boundary with +host-created Git Worktrees, candidate snapshots, and explicit adoption. It will not weaken the +M6a read-only profile. diff --git a/docs/architecture/threat-model.md b/docs/architecture/threat-model.md index 00e1f0f..f2798a8 100644 --- a/docs/architecture/threat-model.md +++ b/docs/architecture/threat-model.md @@ -14,6 +14,7 @@ - Repository files and instructions. - Skills, hooks, and project configuration. - MCP servers and their tool results. +- Child Agent output, summaries, and delegated task results. - Command output and generated patches. ## Initial Controls @@ -83,6 +84,18 @@ snapshots have independent limits. - Server instructions, descriptions, annotations, icons, metadata, stderr, `_meta`, image, audio, and resource content do not enter MCP model-facing Tool contracts or successful results. +- M6a Subagents are created only from immutable host profiles. The complete batch validates unique + bounded tasks and child IDs before child Provider I/O; Providers and Tool executors must be + distinct objects. +- Every M6a child receives a fresh one-message context and exact read-only definitions. Tool + executors must prove governance and `TrustSource.SUBAGENT`; recursive delegation is rejected. +- One `asyncio.TaskGroup` owns all child tasks. Per-child and outer batch deadlines produce typed + ordered results, while external `CancelledError` cancels/joins children and is re-raised. +- Background child `ASK` decisions fail closed in `NON_INTERACTIVE` mode. Parent Policy deny + prevents every child factory and Provider call. +- Child summaries are explicitly untrusted and bounded. Evidence stores ToolCall identity, + error/count metadata, and ToolResult SHA-256 only; Subagent events exclude task, prompt, + message, summary, argument, ToolResult content, repository content, and exception text. ## Non-claims @@ -146,3 +159,13 @@ startup before any Tool Policy decision. - Stdio restricts protocol access to child pipes but not filesystem, network, process, or credential authority. Timeout/termination cannot prove a remote side effect did not complete. +- In-process Subagents isolate Agent message context, not Python memory, operating-system identity, + credentials, Provider access, or malicious host-supplied code. +- M6a read-only admission governs calls through the child executor; it is not an OS sandbox and + cannot prove a host Tool implementation is actually side-effect free. +- Child result/evidence hashes are deterministic equality fingerprints, not signatures, + encryption, provenance, semantic correctness, or durable parent-child audit. +- M6a child deadlines depend on cooperative asyncio cancellation and do not stop arbitrary threads + or prove that an external Provider request incurred no cost. +- M6a does not implement write-capable children, Worktree isolation, candidate adoption, merging, + rollback, durable child Resume, recursive delegation, or token/cost/quality improvements. diff --git a/docs/learning/knowledge-map.md b/docs/learning/knowledge-map.md index 5a54adb..b0641e4 100644 --- a/docs/learning/knowledge-map.md +++ b/docs/learning/knowledge-map.md @@ -816,28 +816,145 @@ ### L11:Subagent 与 Worktree -**理论** +M6 分成两个独立权限阶段: -- Subagent 是有预算的子任务执行者,不允许无限递归。 -- 隔离维度包括上下文、文件、Git 和权限。 -- 返回结果必须携带证据,不只给自然语言结论。 +- **M6a 已实现**:同进程、fresh context、不可递归、只读分析 Subagent; +- **M6b 待实现**:宿主创建 Git Worktree、写入候选快照、单独审批 adoption。 -**Python** +不要因为两者都叫 Subagent 就把“并行读”和“并行改”当成同一个安全问题。 -- 并发任务、取消传播、结构化聚合、进程生命周期。 +**前置知识** -**工程** +- `async def` 返回 coroutine object;只有被 `await` 或包装成 Task 后才执行。 +- coroutine 是计算描述,Task 是事件循环调度和取消的生命周期对象。 +- `asyncio.TaskGroup` 是结构化并发作用域:退出前必须等待所有子 Task 完成或取消,不允许 + 把子任务遗留到父调用之后。 +- `asyncio.Semaphore` 限制同时运行的子任务,不负责结果顺序;结果顺序由 ordinal slot + 单独保证。 +- `asyncio.timeout` 的 deadline cancellation 与外部 `Task.cancel()` 都表现为取消,但外层 + timeout context 只把自己的取消转换成 `TimeoutError`。外部 `CancelledError` 必须继续上抛。 +- Pydantic frozen profile 是宿主能力配置,不是模型建议;`Protocol` 工厂负责创建真实 + Provider 与 Tool executor。 +- Capability profile 与 RBAC/service account 相似:都在执行前固定“可以做什么”;区别是 + 本项目 profile 同时固定 Tool Schema、顺序、Agent 预算、结果预算和 provenance。 +- SHA-256 只能证明两段已观察字节是否一致,不能证明自然语言结论正确,也不是签名、加密或 + 数据来源证明。 + +**核心概念** + +1. **Fresh context,不是 context fork** + + 每个 child 只接收宿主 `system_prompt` 和一个 `Message.user_text(task)`。它不会复制 parent + transcript、parent system prompt、sibling task 或 sibling ToolResult。这样能把子结果归因到 + 一个任务,并降低无关上下文传播。 + +2. **Host profile,不是模型动态角色** + + `SubagentProfile` 固定 parent local Tool name、child system prompt、exact Tool names、 + `AgentLimits`、并发、timeout、summary/evidence/result 上限。模型只能提交 task/reason, + 不能选择 Provider、Tool、权限、深度或 timeout。 + +3. **能力组合先于 Provider I/O** + + `SubagentSupervisor._prepare_children()` 先生成全部唯一 child ID,再创建每个独立 Provider + 和 Tool executor。`validate_child_tools()` 要求 exact ordered names、全部 `READ_ONLY`、 + `governance_enforced is True`,并且每个 Tool 的来源都是 `TrustSource.SUBAGENT`。任一失败使 + 整批在模型调用前关闭。 + +4. **Fan-out/Fan-in 与有序聚合** + + 一个 `TaskGroup` 创建全部 child Task,Semaphore 控制 active concurrency。child 可以乱序 + 完成,但写入固定 ordinal result slot,最终 batch 与输入顺序一致。这类似 Flink keyed + async I/O 中“并发完成 + 按业务序号恢复顺序”,不是按 completion order 直接 append。 + +5. **两层 deadline** + + child timeout 只把一个 child 标成 `TIMED_OUT`,sibling 继续;outer batch timeout 取消 + 所有未完成 Task,并把空 ordinal 补成 `BATCH_TIMED_OUT`。普通 child projection failure + 同样转成 typed result,不触发 fail-fast cancellation。 + +6. **取消是控制信号** + + `CancelledError` 不是普通业务错误。Supervisor 和 parent Tool 都显式 re-raise,让 + TaskGroup 取消并 join 所有 child。若把它吞进 `except Exception` 或转成“执行失败”,父任务 + 会误以为取消成功处理,而后台工作可能继续。 + +7. **Evidence,不是 child 自述** + + `untrusted_summary` 是模型文本,只做字符上限和 NUL 拒绝。真正可核对的是每个 ToolCall 的 + ID、Tool name、error flag、result character count 和 ToolResult UTF-8 SHA-256。原始参数、 + 文件内容和 ToolResult 不复制到 parent evidence/event。 + +8. **后台 ASK 必须 fail closed** + + child executor 固定 `SessionMode.NON_INTERACTIVE`。若 Policy 返回 `ASK`,不会由多个后台 + child 同时弹审批,也不会继承 parent approval,而是直接拒绝。这与 Java 后台 job 不应 + 偷用前台用户会话授权相同。 + +9. **Context isolation 不是 OS sandbox** + + child 仍与 parent 共享 Python 进程、内存、OS 用户、Provider 凭证和宿主 Tool 实现。 + exact read-only Tool profile 约束的是受治理调用路径,不能限制恶意宿主 Python 对象。 + +10. **M6b 为什么必须另做 Worktree** + + 写入会增加 repository identity、dirty base、并发冲突、candidate persistence、adoption + approval、cleanup 和 rollback uncertainty。M6a 不能把“只读 child 已安全”外推成“同进程 + child 可以直接改 parent checkout”。 + +**Java / Flink / Spark 概念映射** + +| 既有经验 | M6a 对应概念 | 关键差异 | +|---|---|---| +| Java `Thread` | `asyncio.Task` | Task 是协作式调度;阻塞代码会卡住同一 event loop | +| `ExecutorService.submit` | `TaskGroup.create_task` | TaskGroup 有词法生命周期,退出前取消/join 全部 child | +| Java 21 `StructuredTaskScope` | `asyncio.TaskGroup` | 都强调 parent-child lifetime;本项目把普通失败投影为 typed result | +| `CompletableFuture.allOf` | fan-out/fan-in | `allOf` 不自动提供本项目的 ordinal aggregation、timeout 分类和结果模型 | +| `Semaphore` / 线程池大小 | `asyncio.Semaphore` | 只限制 active child,不决定输出顺序 | +| Spring Security role/service account | `SubagentProfile` + `TrustSource.SUBAGENT` | profile 还钉住 Tool contract 和资源预算 | +| Flink operator subtask | 独立 child `AgentRuntime` | child 不是分布式进程,没有 checkpoint/restart 隔离 | +| Flink async I/O ordered wait | ordinal result slots | child completion 可乱序,parent batch 仍按输入顺序 | +| Flink cancellation | `CancelledError` propagation | Python 库必须显式不吞取消;线程/外部系统未必协作停止 | +| Spark task result accumulator | `SubagentBatchResult` | summary 不可信,证据只保留有界 metadata/hash | +| 数据血缘 fingerprint | ToolResult SHA-256 | hash 是内容身份,不是业务正确性或来源签名 | + +**代码阅读顺序** -- 限制深度、数量、token、时间和工具权限。 -- 使用 Worktree 隔离可能冲突的修改。 -- 父 Agent 对合并和最终输出负责。 +1. `subagents/models.py`:Profile、Limits、Status、Result 及跨字段约束。 +2. `policy/models.py`:`TrustSource.SUBAGENT` 如何与 MODEL/EXTENSION 区分。 +3. `subagents/contracts.py`:Provider/Tool factory 与 exact read-only capability 校验。 +4. `subagents/evidence.py`:transcript correlation 和 raw-content omission。 +5. `subagents/events.py`:为什么 event union 与 AgentEvent 分开,以及字段刻意缺少什么。 +6. `subagents/supervisor.py`:preflight、TaskGroup、Semaphore、两层 timeout、取消和 ordinal slot。 +7. `subagents/tools.py`:parent JSON Schema、Preview、Policy 接口、canonical result bytes。 +8. `tests/integration/test_governed_subagent_agent.py`:真实 parent/child Agent 与 Read/Search 闭环。 **验收练习** -- 两个只读 Subagent 并行分析独立问题。 -- 子任务超时不阻塞父 Agent。 -- Worktree 不污染主工作区。 -- 冲突时停止并提供 diff,不强行合并。 +1. 从 `SubagentProfile.validate_capabilities()` 开始,列出 child 无法递归 delegation 的两层 + 约束,并解释为什么 builder 还要做跨 profile local-name 冲突检查。 +2. 阅读 `_prepare_children()`,画出 duplicate child ID、Provider factory exception、 + reordered Tool definitions 三种失败分别发生在首个 Provider request 的前还是后。 +3. 对比 `AgentRuntime.run()` 与 `_run_child()`,证明 child first request 只有一个 task + message;写一个测试确保 parent transcript 中的 secret 不出现在 child request。 +4. 用两个 gate Provider 让 ordinal 1 先完成,解释 `result_slots` 为什么仍返回 + `[ordinal 0, ordinal 1]`,并对比按 completion append 的错误实现。 +5. 把 `max_concurrency` 从 2 改成 1,记录 peak active child;说明 Semaphore 为什么不能替代 + TaskGroup 的取消/join 责任。 +6. 制造 child timeout、projection failure、batch timeout 和 parent cancellation,分别记录 + typed result 或抛出异常;解释外部 cancellation 为什么不能转成 `BATCH_TIMED_OUT`。 +7. 给 child Tool 返回包含 secret 的内容,检查 parent evidence/event 只出现 char count/hash; + 再说明 parent Agent Checkpoint 是否可能仍保存完整 ToolResult。 +8. 给 child Policy 添加一条 `ASK` 规则,证明 `NON_INTERACTIVE` 下 approval handler 调用次数为 + 0;说明为什么不能复用 parent 的一次 approval。 +9. 尝试把 `write_file`、`run_command` 或 MCP alias 放进 analysis profile,定位哪一个 + composition invariant 拒绝它,不要只依赖 Tool 名称。 +10. 阅读 `SubagentAnalysisTool._serialize_batch()`,构造 Unicode summary 使 ASCII JSON 膨胀, + 证明上限按最终 UTF-8 bytes 而不是原始 Python 字符数执行。 +11. 检查 `SubagentCompleted` 的字段集合,列出 task、prompt、summary、arguments、ToolResult 和 + exception 为什么都不应进入 observability event。 +12. 为 M6b 写一页威胁清单:clean base、no-checkout Worktree、allowed path、candidate + snapshot、adoption approval、conflict 和 cleanup uncertainty;不要修改 M6a profile。 ### L12:CI、Benchmark 与发布 diff --git a/docs/learning/progress.md b/docs/learning/progress.md index e4c1e47..7cd43ff 100644 --- a/docs/learning/progress.md +++ b/docs/learning/progress.md @@ -13,8 +13,8 @@ | L8 Git/test/repair | Complete and released | M4a Git + M4b Pytest + M4c bounded Repair | | L9 Skills and Hooks | Complete and released | Inert Skills + monotonic Tool Hooks; v0.13 evidence | | L10 MCP | Complete and released | Governed stdio, exact grants, real SDK integration; v0.14 evidence | -| L11 Subagent and Worktree | Not started | | -| L12 CI, benchmark and release | In progress | v0.14 MCP prerelease and cross-platform evidence complete | +| L11 Subagent and Worktree | M6a complete locally; M6b not started | Host-profiled read-only Subagents, TaskGroup, real parent/child integration | +| L12 CI, benchmark and release | In progress | v0.14 released; v0.15 local quality/security gates complete | ## L0 Notes @@ -716,3 +716,72 @@ `1af6a07632abe291ac4adc0ccb04aaa1be5c7d38`。非 draft GitHub prerelease 已发布, 远端 asset name、size 和 GitHub SHA-256 digest 与上述本地 smoke 制品完全一致。 + +## M6a Governed Analysis Subagent Notes + +- M6a 只实现分析 delegation,不实现 child 写入。宿主 `SubagentProfile` 固定 parent local + Tool、child system prompt、exact read-only Tool names、Agent limits、Task/timeout/ + summary/evidence/result budgets;模型只能提交 bounded unique tasks 和 reason。 +- child 不是 parent context fork。每个 `AgentRuntime` 从一个 fresh user message 开始, + 不继承 parent/sibling transcript,因此 context attribution 更清晰,但不能据此宣称 + OS、内存、Provider 凭证或数据隔离。 +- 所有 child ID 在工厂调用前一次性生成并验证唯一。Provider 和 governed Tool executor + 必须逐 child 独立,Tool definition 顺序必须与 profile 完全一致,全部为 `READ_ONLY`, + `governance_enforced is True`,且 provenance 为 `TrustSource.SUBAGENT`。 +- profile 拒绝任何 `delegate_` child Tool;`build_subagent_tools()` 再拒绝 duplicate + profile ID/local name 和跨 profile parent-local/child-name 冲突,避免形成递归能力图。 +- 一个 `asyncio.TaskGroup` 拥有全部 child Task;Semaphore 只限制 active concurrency, + ordinal slots 单独保证 output order。代码没有 detached Task、daemon thread 或后台进程。 +- child timeout/failure 只生成该 ordinal 的 typed result,sibling 继续;outer batch timeout + 取消未完成 Task 并填充 `BATCH_TIMED_OUT`。外部 `CancelledError` 在 child、Supervisor 和 + parent Tool 路径显式 re-raise。 +- child 固定 `NON_INTERACTIVE`,因此 Policy `ASK` 不会弹出嵌套审批,而是 fail closed。 + parent delegation Policy 与 child Tool Policy 是两次独立判断。 +- `untrusted_summary` 只做长度/NUL边界,不是证据。Evidence 只保留 ToolCall ID/name、 + error、content character count 与 ToolResult UTF-8 SHA-256,不复制参数或原始结果。 +- Subagent events 只含 parent call ID、profile、child/ordinal、status、duration、counts、 + usage 和 result hash;task、prompt、messages、summary、arguments、ToolResult、repository + content 和 exception text 均不进入 event。 +- parent `SubagentAnalysisTool` 生成 profile-specific Draft 2020-12 Schema、medium-risk + preview 和 canonical ASCII result;最终按 UTF-8 byte budget 拒绝 oversized batch。 +- M6a 未做 token/latency/cost/quality benchmark。额外 child Provider/Tool 调用可能降低 + parent context pressure,也可能增加成本与尾延迟,简历不能写“节省 X% token”。 + +## M6a Review Lessons + +- “最多并发 N 个”不能只靠创建 N 个 detached Task;并发度和生命周期所有权是两个问题。 + Semaphore 解决前者,TaskGroup 解决后者。 +- `CancelledError` 是控制流,不是普通失败。把它转成 child failed 会让调用方无法区分用户 + 取消与业务错误,并破坏父子任务终止语义。 +- preflight 不只是 Schema。重复 child ID 如果直到 result model 才发现,Provider 已产生 + 成本,且两个 child 会共享 run ID;因此 ID tuple 必须先完整验证。 +- “只读 profile”必须验证 definition side effect、governance marker 和 provenance,不能靠 + Tool 名称或 prompt 约定。 +- parent Tool 结果再次验证 `SubagentBatchResult`,并在最后一步按 serialized bytes 计预算; + Python 字符数无法代表 escaped ASCII JSON 大小。 +- Pytest 默认 import mode 会把非 package 测试目录中的同名文件当成同一顶层模块。新增 + `tests/unit/subagents/__init__.py` 后,完整 suite 才能与既有 `test_events.py`、 + `test_models.py`、`test_tools.py` 同时收集。 +- Worktree 能隔离 checkout 路径冲突,但不是 OS sandbox,也不自动解决 candidate adoption、 + conflict、cleanup、rollback 或 concurrent host mutation;这些必须留给 M6b 单独设计。 + +## M6a Local Verification + +- 完整 parent/child 集成使用真实 `AgentRuntime`、`GovernedToolExecutor`、 + `ReadFileTool`/`SearchTextTool` 和 `WorkspaceBoundary`,仅 Provider 采用确定性脚本。 +- 集成证明 parent 只产生一个 delegation ToolCall、两个 child context 各只有一个 fresh + task message、read/search evidence hash 存在、child provenance 为 SUBAGENT、parent + batch ordinal 有序、event 无 task/prompt/content,且 Workspace 前后 bytes 完全一致。 +- deny case 在任何 Provider/Tool factory 前停止;递归 ToolCall 得到 `unknown_tool`;一个 + child timeout 不影响 sibling/parent;parent cancellation 取消两个 blocked child 且 active + count 回到 0。 +- 安全审查增加 duplicate child ID 与 duplicate direct task 红灯回归,修复后完整 + Subagent + integration suite 为 100 passed。 +- Python 3.12.13 在最终两条 hardening 测试加入前为 1060 passed、10 skipped、91.08% + branch coverage;最终 Python 3.13.14 为 1062 passed、10 skipped、91.09%。skip 均来自 + 当前 Windows 会话缺少 symlink privilege。 +- Ruff format/check、strict Pyright、Bandit 与 locked runtime pip-audit 已通过; + pip-audit 为 `No known vulnerabilities found`。 +- `0.15.0a0` version contract 和 installed-package Subagent API smoke 已通过。最终双版本 + release rerun、reproducible artifact、GitHub PR/CI/tag/Release 证据留给 Task 9,当前不 + 宣称已发布。 diff --git a/docs/resume/project-profile.md b/docs/resume/project-profile.md index 7447312..b4d1bc2 100644 --- a/docs/resume/project-profile.md +++ b/docs/resume/project-profile.md @@ -4,12 +4,12 @@ > Registry、M2b 受治理文件写入、M2c argv 命令执行与 M3a 确定性 Context Budget > 、M3b 版本化 Session/追加式 Trace、M3c Checkpoint/Resume、M4a hardened 只读 Git > 、M4b 受治理 Pytest 诊断、M4c 宿主控制的有限 Repair 及 M5a 惰性 Skills/Tool Hooks -> 与 M5b host-pinned local stdio MCP 已发布。Shell 字符串、 -> 项目可执行 Hook、OS 沙箱、remote HTTP/OAuth MCP、自动 Repair Resume、Subagent/ -> Worktree 和真实凭证联调尚未实现。`v0.14.0-alpha.0` GitHub prerelease 已发布; -> Python 3.12/3.13 本地各 961 passed、10 个 Windows symlink 条件跳过、90.84% -> 分支覆盖率,Ruff/Pyright/Bandit/locked pip-audit、四组 artifact smoke、PR/main -> 跨平台 CI、tag 与 Release asset digest 验证均已通过。 +> 、M5b host-pinned local stdio MCP 已发布,M6a host-profiled read-only analysis +> Subagent 已完成本地实现与安全门。Shell 字符串、项目可执行 Hook、OS 沙箱、remote +> HTTP/OAuth MCP、自动 Repair Resume、write-capable Subagent/Worktree 和真实凭证联调 +> 尚未实现。`v0.14.0-alpha.0` GitHub prerelease 已发布;`0.15.0a0` 当前最终本地 +> Python 3.13 为 1062 passed、10 个 Windows symlink 条件跳过、91.09% 分支覆盖率, +> Python 3.12 的最终 release rerun、artifact、CI、tag 和 Release 尚待执行。 > > 本文中的功能、性能和指标是目标或验收方案。只有得到代码、测试、CI、Benchmark 或 Release 证据后,才能改写为已完成成果。 @@ -23,7 +23,15 @@ M5a 进一步加入 source-qualified Skill Catalog:严格解析 `SKILL.md`, 不能改写结果。M5b 再接入官方稳定 MCP Python SDK 的 local stdio Tools:启动前审批绝对 executable/argv/cwd,钉住 server identity、完整 Tool 集合和 input/output schema hash, 用 owner-worker 管理跨 Task 进程生命周期,并把 MCP alias 作为 extension 继续送入同一 -Policy/approval/result-boundary。下一阶段将实现受限 Subagent 与 Git Worktree。 +Policy/approval/result-boundary。下一阶段将实现 write-capable Subagent 的 Git Worktree +candidate/adoption 边界。 + +M6a 已加入受限分析 Subagent:宿主以 immutable profile 固定 child system prompt、exact +read-only Tool、Agent/timeout/result budgets 和 `TrustSource.SUBAGENT`;每个 child 使用 +fresh context,所有 Task 归属一个 `asyncio.TaskGroup`,通过 Semaphore、per-child/outer +timeout、ordinal slots 完成有界并发与有序聚合。Parent 只接收 labelled untrusted summary +和 ToolResult metadata/SHA-256 evidence;事件不记录 task/prompt/arguments/results。 +M6b 才会设计 Git Worktree candidate/adoption,M6a 不具备写入和合并权限。 ## 2. 项目定位 @@ -38,7 +46,8 @@ Policy/approval/result-boundary。下一阶段将实现受限 Subagent 与 Git W 最终技术栈以 `pyproject.toml`、ADR 和发布版本为准。 -M0 至 M5b 已实际使用 Python 3.12/3.13、`asyncio`、`Protocol`、`dataclasses`、uv、 +M0 至 M6a 已实际使用 Python 3.12/3.13、`asyncio`、`asyncio.TaskGroup`、 +`asyncio.Semaphore`、`Protocol`、`dataclasses`、uv、 Hatchling、Pydantic v2、pydantic-settings、Platformdirs、HTTPX、httpx-sse、Typer、Rich、 JSON Schema Draft 2020-12、stdlib `sqlite3`、SQLite WAL/事务/索引、canonical JSON、SHA-256、 Git porcelain v2、Pytest/JUnit XML、defusedxml、PyYAML、官方 MCP Python SDK v1、 @@ -60,6 +69,7 @@ JSON-RPC/stdio、pytest-asyncio、Coverage、Ruff 与 Pyright;其余技术随 | 测试诊断与修复 | 固定 Pytest Profile、JUnit XML、双状态分类、有限 Repair 状态机、失败指纹、多维预算 | | 扩展治理 | restricted PyYAML、source-qualified Skill Catalog、SHA/文件身份重验、typed async Tool Hooks、monotonic authorization | | MCP 互操作 | 官方 `mcp` SDK v1、JSON-RPC、local stdio、host-pinned grants、canonical schema SHA-256、owner-worker lifecycle | +| Subagent 编排 | immutable capability profile、fresh context、`asyncio.TaskGroup`/Semaphore、结构化取消、SUBAGENT provenance、ordered aggregation、evidence SHA-256 | | 测试与质量 | Pytest、pytest-asyncio、Coverage、Ruff、Pyright | | 构建与发布 | `uv`、`pyproject.toml`、GitHub Actions、SemVer、GitHub Release | | 文档与治理 | Markdown、ADR、威胁模型、贡献指南、Changelog | @@ -80,7 +90,8 @@ JSON-RPC/stdio、pytest-asyncio、Coverage、Ruff 与 Pyright;其余技术随 12. 企业级质量门禁与发布工程。 13. 惰性不可信 Skills 与单调授权 Tool Hooks。 14. Host-pinned、双审批、受治理的 MCP stdio Tools。 -15. 面向 Subagent 与 Worktree 的扩展架构。 +15. 宿主能力限定、结构化并发、不可递归的只读分析 Subagent。 +16. 面向 Git Worktree candidate/adoption 的后续扩展架构。 ## 5. 亮点拆解 @@ -105,8 +116,9 @@ JSON-RPC/stdio、pytest-asyncio、Coverage、Ruff 与 Pyright;其余技术随 | 宿主控制的有限 Repair Loop | 诊断可用后,若让模型自行决定改什么、何时测试和是否重试,会形成无界反馈、覆盖用户改动或用文字冒充成功 | 独立 `RepairRuntime`、clean repository、literal exact tracked scope、pre-policy `RepairActionGuard`、固定 Pytest、Git/Workspace 前后证据、canonical failure fingerprint、attempt/time/patch/prompt/repeated-failure 预算、SQLite schema v3 Repair hash chain | 先跑 baseline;每轮只允许一次 Agent read/edit;宿主验证 patch 和测试无副作用后重测,只在完整 passing evidence 下成功,否则以 typed reason 停止 | 阻止 scope 外写入、execute/network、自声明成功、staged/untracked/ignored/submodule/branch 漂移、重复失败和测试残留修改;中断会话不自动重放 | 真实集成覆盖一次缺陷修复、越权写入在审批/落盘前拒绝、dirty repo 在 Provider/Pytest 前拒绝、测试修改仓库即使通过也停止;Python 3.12/3.13 本地各 798 passed、6 个 Windows symlink 条件跳过,3.13 分支覆盖率 90.88%;未虚构 benchmark 提升率 | | 惰性 Skills 与单调授权 Hooks | 仓库扩展既会占用上下文,也可能通过静默覆盖、动态导入或生命周期回调绕过权限 | restricted PyYAML/Pydantic、direct-child regular-file/reparse 检查、source-qualified ID、SHA-256 + file identity TOCTOU 重验、只读 list/load Tool、async Protocol Hook、稳定优先级、timeout、bounded audit | 模型先发现 metadata,再按 fingerprint 加载 labelled untrusted Markdown;宿主 pre-Hook 可 veto,post-Hook 可观察 | 阻止 Skill 注册执行能力、跨来源 shadow、内容漂移、无界扫描和 Hook 提权;pre 失败在副作用前关闭,post 失败不伪造执行事实 | 真实 Agent 证明恶意 Skill 不能绕过 deny、pre 阻断零落盘、post 失败后结果与后续 observer 保留;Python 3.12/3.13 各 867 passed、90.86% 分支覆盖率;PR/main 五 job CI、v0.13 prerelease 与远端制品摘要验证通过 | | 受治理 MCP stdio | 直接信任 `tools/list` 会让 server/package 替换新增权限;local server 在 Tool Policy 前已能执行代码,连接审批也不能代表每次调用获批 | 官方 SDK `mcp>=1.28.1,<2`、absolute executable + argv-only、SecretStr environment、独立 connection approver、protocol/server identity、exact grant set、canonical input/output schema hash、owner-worker、per-tool `TrustSource.EXTENSION`、bounded result validator | 宿主审核一个固定 local stdio server,验证后只发布 local aliases;每次调用继续经过 Schema/Preview/Hook/Policy/Tool approval,返回 text/structured JSON | 阻止 PATH/shell 注入、未授权 Tool、schema drift、server metadata 提权、跨 Task AnyIO context 泄漏、无界/多媒体结果和 approval 混淆;超时保留副作用不确定性 | 真实官方 SDK 进程覆盖 handshake/call/shutdown、deny/ask 零远端调用、extra Tool/schema drift 零 admission、cross-task close;本地 Python 3.12/3.13 各 961 passed/10 skips、90.84% branch coverage;PR/main 五 job CI、四组 artifact smoke、v0.14 prerelease 与远端制品摘要验证通过 | +| 受限分析 Subagent | 单 Agent 串行探索会把独立调查混入 parent transcript;直接复制 parent 权限又会放大 Tool、成本、递归和取消风险 | immutable `SubagentProfile`、独立 Provider/Tool factory、fresh child `AgentRuntime`、exact read-only definitions、`TrustSource.SUBAGENT`、`SessionMode.NON_INTERACTIVE`、`asyncio.TaskGroup` + Semaphore、per-child/outer timeout、ordinal slots、canonical result/evidence SHA-256 | Parent 经一个受治理 Tool fan-out 1-4 个独立分析任务;child 乱序完成但按输入顺序聚合,单 child timeout/failure 不影响 sibling,parent cancellation 取消并 join 全部 child | 防止 parent/sibling context 隐式泄漏、模型动态扩权、递归 delegation、后台嵌套审批、orphan Task、原始 ToolResult 进入 parent/event;将 child 自述与可核对 hash metadata 分开 | 真实 parent/child Agent + Read/Search 集成覆盖 fresh context、SUBAGENT Policy、deny 零 factory、non-recursion、timeout isolation、Workspace bytes 不变和双 child cancellation;Subagent/integration 100 passed,最终 Python 3.13 全套 1062 passed/10 skips、91.09% branch coverage;未宣称 token/latency 提升 | | 质量门禁 | 企业级项目需要稳定接口和回归保护 | Ruff、严格 Pyright、Pytest、85% 核心覆盖率门槛、哈希构建约束、CI、SemVer | 自动执行 lint、类型检查、测试、构建和安装验证 | 防止低质量变更进入发布版本 | v0.12:Python 3.12/3.13 本地各 798 通过、6 项 symlink 条件跳过,90.88% 分支覆盖率,Bandit/pip-audit 与四组 artifact smoke 通过;PR/main CI 的 Ubuntu/Windows × 3.12/3.13 与 quality 全成功,prerelease 及两个校验摘要一致的制品已发布 | -| 可扩展 Harness | Skills、Hooks、MCP、Subagent 会增加控制流复杂度 | 稳定 Protocol、EventBus、能力声明、依赖倒置、per-tool provenance | 在不侵入 Agent Core 的前提下加入 Skills、Hooks 与 MCP | 避免扩展绕过权限、Trace 和 Session | Skills/Hooks/MCP 均复用 Tool Registry 与 Policy;Subagent/Worktree 待实现 | +| 可扩展 Harness | Skills、Hooks、MCP、Subagent 会增加控制流复杂度 | 稳定 Protocol、EventBus、能力声明、依赖倒置、per-tool provenance | 在不侵入 Agent Core 的前提下加入 Skills、Hooks、MCP 与只读 Subagent | 避免扩展绕过权限、Trace 和 Session | Skills/Hooks/MCP/Subagent 均复用 Tool Registry 与 Policy;write-capable Worktree 待实现 | ## 6. 指标回填规则 @@ -184,15 +196,26 @@ input/output schema hash,任何 extra/missing/schema drift 都零 admission。 context 由 owner worker 进入和退出,解决跨 Task close;result 只接受有界 text/structured JSON。stdio 和审批都不是 sandbox,timeout 也只能报告副作用完成状态未知。” -### 7.11 企业级体现在哪里 +### 7.11 Subagent 为什么不是“开几个 asyncio Task” + +“并发只是表面,真正的问题是能力、上下文、生命周期和证据。我用宿主 immutable profile +固定每个 child 的 system prompt、exact read-only Tools、Agent/timeout/result budgets, +并要求 Tool executor 的 provenance 为 `SUBAGENT`;child 只拿一个 fresh task message, +不能继承 parent transcript 或递归 delegation。所有 child Task 都在一个 TaskGroup 中, +Semaphore 限并发、ordinal slot 保序、child/batch timeout 分层,外部 cancellation 原样 +传播并 join 全部 child。返回的 summary 明确不可信,证据只保留 ToolResult hash metadata。 +这不是 OS sandbox,也没有 benchmark 证明 token 或延迟改善。” + +### 7.12 企业级体现在哪里 “企业级不是功能数量,而是边界清晰、失败可诊断、状态可恢复、安全策略可测试、发布可重复。项目设置严格类型、测试覆盖率门槛、跨平台 CI、安全模型、SemVer 和发布 smoke test。” -### 7.12 如何避免过度设计 +### 7.13 如何避免过度设计 -“首版先完成单 Agent 的最小完整闭环。Skills、Hooks 和 MCP 已沿已有 Tool、Event、 -Policy、Session 协议接入;Subagent 和 Worktree 也必须复用这些边界,不能绕过权限与 -Trace。remote MCP/OAuth 等独立威胁面不与 local stdio 混做。” +“首版先完成单 Agent 的最小完整闭环。Skills、Hooks、MCP 和 M6a read-only Subagent +都沿已有 Tool、Event、Policy、Session 协议接入;write-capable Subagent/Worktree 要等 +candidate/adoption 边界单独实现,不能把只读 profile 直接放宽。remote MCP/OAuth 等独立 +威胁面也不与 local stdio 混做。” ## 8. 简历成果模板 @@ -257,6 +280,12 @@ Trace。remote MCP/OAuth 等独立威胁面不与 local stdio 混做。” protocol/server identity、exact Tool grant set 与 input/output schema hash;通过 owner-worker 管理官方 SDK 跨 Task 生命周期,并让 alias 以 `TrustSource.EXTENSION` 继续经过 Policy/approval 和有界结果校验。 +- 实现宿主能力限定的只读分析 Subagent:以 fresh context、exact Tool set、 + `TrustSource.SUBAGENT`、TaskGroup/Semaphore、双层 timeout 和 ordinal aggregation + 并行处理 1-4 个独立任务;普通 child 失败隔离,parent cancellation 无 orphan Task。 +- 以 labelled untrusted summary 和 ToolResult metadata/SHA-256 聚合 child 结果,事件 + 排除 task/prompt/message/arguments/results/exception;真实 parent/child Read/Search + 集成验证 Policy deny 零工厂调用、不可递归、Workspace bytes 不变和取消传播。 - Python 3.12/3.13 各 678 项通过、5 项因 Windows symlink 权限跳过,分支覆盖率 90.25%;Bandit/pip-audit 与 wheel/sdist 四组隔离安装 smoke 通过。 - 完成 Mini CodeAgent M0 工程基础:显式配置优先级、Pydantic 强类型边界、密钥安全 JSON 日志与 `doctor` 诊断 CLI。 diff --git a/docs/superpowers/plans/2026-07-01-m6a-bounded-subagents.md b/docs/superpowers/plans/2026-07-01-m6a-bounded-subagents.md index 492dcdc..80d11fc 100644 --- a/docs/superpowers/plans/2026-07-01-m6a-bounded-subagents.md +++ b/docs/superpowers/plans/2026-07-01-m6a-bounded-subagents.md @@ -63,7 +63,7 @@ ToolRegistry, GovernedToolExecutor, Policy/Hooks, Pytest/pytest-asyncio, Ruff, s - Create: `src/mini_code_agent/subagents/models.py` - Create: `tests/unit/subagents/test_models.py` -- [ ] **Step 1: Write failing provenance tests** +- [x] **Step 1: Write failing provenance tests** Add: @@ -102,7 +102,7 @@ def test_rule_can_match_subagent_without_allowing_parent_model() -> None: assert parent.rule_id == "default-read-only" ``` -- [ ] **Step 2: Run the provenance tests and observe red** +- [x] **Step 2: Run the provenance tests and observe red** Run: @@ -112,7 +112,7 @@ py -m uv run pytest tests/unit/policy/test_engine.py -q Expected: the stable enum assertion fails because `subagent` does not exist. -- [ ] **Step 3: Add the public provenance value** +- [x] **Step 3: Add the public provenance value** Add exactly: @@ -127,7 +127,7 @@ class TrustSource(StrEnum): Run the focused Policy tests and expect all to pass. -- [ ] **Step 4: Write failing profile/model tests** +- [x] **Step 4: Write failing profile/model tests** Cover: @@ -193,7 +193,7 @@ Also reject task/result/status IDs over bounds, non-static public errors, summar characters, evidence over 256 items, `max_tasks > 4`, `max_concurrency > 4`, child timeout over 600, batch timeout over 900, and batch timeout lower than child timeout. -- [ ] **Step 5: Run model tests and observe collection failure** +- [x] **Step 5: Run model tests and observe collection failure** Run: @@ -203,7 +203,7 @@ py -m uv run pytest tests/unit/subagents/test_models.py -q Expected: import failure because the package does not exist. -- [ ] **Step 6: Implement immutable models** +- [x] **Step 6: Implement immutable models** Implement these stable public shapes: @@ -266,7 +266,7 @@ Add bounded `SubagentEvidenceItem`, `SubagentChildResult`, `SubagentBatchResult` enum, and runtime error class. Freeze tuples and calculate count consistency in a model validator. Do not include raw task, prompt, arguments, ToolResult content, or exception text. -- [ ] **Step 7: Run model, Policy, Ruff, and Pyright checks** +- [x] **Step 7: Run model, Policy, Ruff, and Pyright checks** Run: @@ -278,7 +278,7 @@ py -m uv run pyright src/mini_code_agent/subagents/models.py tests/unit/subagent Expected: all pass. -- [ ] **Step 8: Commit provenance and models** +- [x] **Step 8: Commit provenance and models** ```powershell git add src/mini_code_agent/policy/models.py src/mini_code_agent/subagents/__init__.py src/mini_code_agent/subagents/models.py tests/unit/policy/test_engine.py tests/unit/subagents/test_models.py @@ -293,7 +293,7 @@ git commit -m "feat: define bounded subagent profiles" - Create: `tests/unit/subagents/test_contracts.py` - Create: `tests/unit/subagents/test_evidence.py` -- [ ] **Step 1: Write failing composition tests** +- [x] **Step 1: Write failing composition tests** Build fake factories and prove: @@ -321,7 +321,7 @@ def test_validate_child_tools_rejects_authority_drift( Test that Provider and Tool factories receive only profile, child ID, and pinned workspace root; their exceptions map to one static composition error. -- [ ] **Step 2: Write failing evidence tests** +- [x] **Step 2: Write failing evidence tests** Construct a real `AgentResult` transcript with two ToolCalls and correlated ToolResults: @@ -347,7 +347,7 @@ Also reject duplicate/missing correlation, result-before-call, excessive items, roles, and mismatched child result IDs. Verify canonical hashes ignore dictionary insertion order and reject NaN/Infinity. -- [ ] **Step 3: Run tests and observe red** +- [x] **Step 3: Run tests and observe red** ```powershell py -m uv run pytest tests/unit/subagents/test_contracts.py tests/unit/subagents/test_evidence.py -q @@ -355,7 +355,7 @@ py -m uv run pytest tests/unit/subagents/test_contracts.py tests/unit/subagents/ Expected: missing modules/functions. -- [ ] **Step 4: Implement composition protocols and validation** +- [x] **Step 4: Implement composition protocols and validation** Define: @@ -387,7 +387,7 @@ class GovernedSubagentTools(ToolExecutor, Protocol): `governance_enforced is True`, and `trust_source_for(name) is SUBAGENT`. Catch implementation exceptions and raise only `SubagentCompositionError`. -- [ ] **Step 5: Implement evidence extraction and canonical hash** +- [x] **Step 5: Implement evidence extraction and canonical hash** Walk messages in order. Register each assistant ToolCall exactly once; accept a user ToolResult only after its call; emit one item in ToolCall order; require every call to have one result. Hash UTF-8 @@ -408,7 +408,7 @@ def subagent_result_sha256(value: BaseModel) -> str: ... Use canonical ASCII JSON with sorted keys, compact separators, `allow_nan=False`, and SHA-256. -- [ ] **Step 6: Run focused and static checks** +- [x] **Step 6: Run focused and static checks** ```powershell py -m uv run pytest tests/unit/subagents/test_contracts.py tests/unit/subagents/test_evidence.py -q @@ -416,7 +416,7 @@ py -m uv run ruff check src/mini_code_agent/subagents tests/unit/subagents py -m uv run pyright src/mini_code_agent/subagents tests/unit/subagents ``` -- [ ] **Step 7: Commit contracts and evidence** +- [x] **Step 7: Commit contracts and evidence** ```powershell git add src/mini_code_agent/subagents/contracts.py src/mini_code_agent/subagents/evidence.py tests/unit/subagents/test_contracts.py tests/unit/subagents/test_evidence.py @@ -431,7 +431,7 @@ git commit -m "feat: validate subagent capabilities and evidence" - Create: `tests/unit/subagents/test_events.py` - Create: `tests/unit/subagents/test_supervisor.py` -- [ ] **Step 1: Write failing event tests** +- [x] **Step 1: Write failing event tests** Require immutable bounded events: @@ -472,7 +472,7 @@ def test_subagent_events_omit_task_prompt_summary_and_results() -> None: Round-trip all four event types through `TypeAdapter[SubagentEvent]`; reject IDs, durations, counts, and hashes over bounds; prove sink exceptions do not alter execution. -- [ ] **Step 2: Write failing one-child supervisor tests** +- [x] **Step 2: Write failing one-child supervisor tests** Use injected deterministic ID and monotonic factories: @@ -508,19 +508,19 @@ async def test_one_child_gets_fresh_context_exact_tools_and_bounded_result( Also test a non-completed `StopReason`, provider factory error, Tool factory error, duplicate object identity, malformed AgentResult, and event order. -- [ ] **Step 3: Run tests and observe red** +- [x] **Step 3: Run tests and observe red** ```powershell py -m uv run pytest tests/unit/subagents/test_events.py tests/unit/subagents/test_supervisor.py -q ``` -- [ ] **Step 4: Implement Subagent events** +- [x] **Step 4: Implement Subagent events** Keep the union separate from `AgentEvent` because parent run/turn context is unavailable at the Tool boundary. Add `NullSubagentEventSink` and `RecordingSubagentEventSink`. Publishing catches ordinary sink exceptions; cancellation is not involved because publishing is synchronous. -- [ ] **Step 5: Implement child preparation and execution** +- [x] **Step 5: Implement child preparation and execution** `SubagentSupervisor.__init__` snapshots profile/factories/root and validates the root. Before any Provider call, `_prepare_children()` creates all providers/executors, rejects reused object IDs, @@ -540,7 +540,7 @@ result = await runtime.run( Map completed, stopped, timed-out, and failed paths to static result models. Truncate summaries by profile character limit before hashing. Never copy `AgentResult.messages` into a public model. -- [ ] **Step 6: Run focused and static checks** +- [x] **Step 6: Run focused and static checks** ```powershell py -m uv run pytest tests/unit/subagents/test_events.py tests/unit/subagents/test_supervisor.py -q @@ -548,7 +548,7 @@ py -m uv run ruff check src/mini_code_agent/subagents tests/unit/subagents py -m uv run pyright src/mini_code_agent/subagents tests/unit/subagents ``` -- [ ] **Step 7: Commit the one-child lifecycle** +- [x] **Step 7: Commit the one-child lifecycle** ```powershell git add src/mini_code_agent/subagents/events.py src/mini_code_agent/subagents/supervisor.py tests/unit/subagents/test_events.py tests/unit/subagents/test_supervisor.py @@ -561,7 +561,7 @@ git commit -m "feat: run isolated subagent lifecycles" - Modify: `src/mini_code_agent/subagents/supervisor.py` - Modify: `tests/unit/subagents/test_supervisor.py` -- [ ] **Step 1: Write failing concurrency tests** +- [x] **Step 1: Write failing concurrency tests** Add deterministic gate providers: @@ -599,13 +599,13 @@ Add: - no pending task remains after completion/cancellation; - completed/timed-out/failed counts and batch hash are deterministic. -- [ ] **Step 2: Run concurrency tests and observe red** +- [x] **Step 2: Run concurrency tests and observe red** ```powershell py -m uv run pytest tests/unit/subagents/test_supervisor.py -q -k "concurrent or timeout or cancel" ``` -- [ ] **Step 3: Implement TaskGroup fan-out/fan-in** +- [x] **Step 3: Implement TaskGroup fan-out/fan-in** Use one result slot per input ordinal and one semaphore: @@ -631,7 +631,7 @@ except TimeoutError: `_run_child()` wraps only its own timeout and ordinary failures. It must always re-raise `CancelledError`. External cancellation must not be converted into batch timeout. -- [ ] **Step 4: Verify focused and complete Subagent tests** +- [x] **Step 4: Verify focused and complete Subagent tests** ```powershell py -m uv run pytest tests/unit/subagents -q @@ -639,7 +639,7 @@ py -m uv run ruff check src/mini_code_agent/subagents tests/unit/subagents py -m uv run pyright src/mini_code_agent/subagents tests/unit/subagents ``` -- [ ] **Step 5: Commit structured concurrency** +- [x] **Step 5: Commit structured concurrency** ```powershell git add src/mini_code_agent/subagents/supervisor.py tests/unit/subagents/test_supervisor.py @@ -653,7 +653,7 @@ git commit -m "feat: coordinate bounded subagent batches" - Create: `tests/unit/subagents/test_tools.py` - Modify: `src/mini_code_agent/subagents/__init__.py` -- [ ] **Step 1: Write failing Tool tests** +- [x] **Step 1: Write failing Tool tests** Cover: @@ -691,13 +691,13 @@ oversized serialized result, supervisor failure, malformed result, and builder c Prove `build_subagent_tools()` rejects duplicate profile IDs/local names and any parent local name appearing in any child `tool_names`. -- [ ] **Step 2: Run Tool tests and observe red** +- [x] **Step 2: Run Tool tests and observe red** ```powershell py -m uv run pytest tests/unit/subagents/test_tools.py -q ``` -- [ ] **Step 3: Implement dynamic Tool definitions** +- [x] **Step 3: Implement dynamic Tool definitions** The input schema is profile-specific: @@ -726,12 +726,12 @@ Snapshot a distinct `ToolDefinition` per profile. Preview never exposes task tex to the supervisor, serializes ASCII canonical JSON, checks UTF-8 byte length, and returns static errors. Re-raise cancellation. -- [ ] **Step 4: Export the stable M6a API** +- [x] **Step 4: Export the stable M6a API** Export profile/limits/status/results/evidence/errors, factory protocols, event models/sinks, supervisor, Tool, and builder. Do not export internal prepared-child or transcript walkers. -- [ ] **Step 5: Verify Tool, Registry, and Policy paths** +- [x] **Step 5: Verify Tool, Registry, and Policy paths** ```powershell py -m uv run pytest tests/unit/subagents/test_tools.py tests/unit/tools/test_registry.py tests/unit/policy -q @@ -739,7 +739,7 @@ py -m uv run ruff check src/mini_code_agent/subagents tests/unit/subagents py -m uv run pyright src/mini_code_agent/subagents tests/unit/subagents ``` -- [ ] **Step 6: Commit the parent adapter** +- [x] **Step 6: Commit the parent adapter** ```powershell git add src/mini_code_agent/subagents tests/unit/subagents/test_tools.py @@ -751,7 +751,7 @@ git commit -m "feat: expose governed analysis subagents" **Files:** - Create: `tests/integration/test_governed_subagent_agent.py` -- [ ] **Step 1: Build real governed child Tools** +- [x] **Step 1: Build real governed child Tools** Create a test `SubagentToolFactory` that returns a new `GovernedToolExecutor` per child: @@ -774,7 +774,7 @@ def create( Assert exact profile names equal the produced definitions. -- [ ] **Step 2: Write the parent Agent integration** +- [x] **Step 2: Write the parent Agent integration** Use a parent `ScriptedProvider` that calls `delegate_analysis` with two tasks, then stops. Use a factory that returns two child scripted Providers; each child calls a real read-only Tool and then @@ -790,7 +790,7 @@ Assert: - child Tool trust source is `SUBAGENT`; - parent and child Workspaces remain byte-identical. -- [ ] **Step 3: Add deny, timeout, and non-recursion integration cases** +- [x] **Step 3: Add deny, timeout, and non-recursion integration cases** Prove: @@ -799,7 +799,7 @@ Prove: - one timed-out child does not stop its sibling; - parent task cancellation cancels both children. -- [ ] **Step 4: Run integration and leak assertions** +- [x] **Step 4: Run integration and leak assertions** ```powershell py -m uv run pytest tests/integration/test_governed_subagent_agent.py -q @@ -809,7 +809,7 @@ rg -n "task text|system_prompt|arguments|ToolResult content|exception" src/mini_ Expected: all tests pass; the event scan shows no payload fields. -- [ ] **Step 5: Commit integration evidence** +- [x] **Step 5: Commit integration evidence** ```powershell git add tests/integration/test_governed_subagent_agent.py @@ -821,7 +821,7 @@ git commit -m "test: prove governed subagent delegation" **Files:** - Modify only files implicated by a failing regression test. -- [ ] **Step 1: Run full branch coverage** +- [x] **Step 1: Run full branch coverage** Run on Python 3.12 and 3.13: @@ -834,7 +834,7 @@ py -m uv run --no-sync pytest --cov -q Record exact pass/skip counts and branch coverage. -- [ ] **Step 2: Run static, security, and dependency gates** +- [x] **Step 2: Run static, security, and dependency gates** ```powershell py -m uv run --no-sync ruff format --check . @@ -845,7 +845,7 @@ py -m uv export --locked --no-dev --no-emit-project --format requirements.txt -o py -m uv tool run --python 3.13 pip-audit -r build/runtime-requirements.txt ``` -- [ ] **Step 3: Inspect trust and concurrency boundaries** +- [x] **Step 3: Inspect trust and concurrency boundaries** ```powershell git diff main...HEAD -- src/mini_code_agent/subagents src/mini_code_agent/policy/models.py @@ -862,7 +862,7 @@ Confirm: - no raw task, prompt, argument, result, or exception enters events/evidence; - no recursive local Tool is admitted. -- [ ] **Step 4: Fix every issue with red-green regression** +- [x] **Step 4: Fix every issue with red-green regression** For each issue, add the smallest failing test, run it to observe red, apply the focused fix, rerun the focused and full Subagent suites, and commit: @@ -889,13 +889,13 @@ If no issue is found, do not create an empty commit. - Modify: `uv.lock` - Modify: release-version tests and `tests/smoke_test.py` -- [ ] **Step 1: Write architecture, ADR, and threat model** +- [x] **Step 1: Write architecture, ADR, and threat model** Document fresh child context, exact host profile, no recursion, SUBAGENT provenance, TaskGroup lifetime, per-child/outer timeout, cancellation, evidence hashes, event omissions, and why in-process children are not an OS sandbox. -- [ ] **Step 2: Add L11 prerequisites and exercises** +- [x] **Step 2: Add L11 prerequisites and exercises** Teach: @@ -909,12 +909,12 @@ Teach: Add at least eight code-reading exercises tied to exact M6a modules. -- [ ] **Step 3: Update resume material** +- [x] **Step 3: Update resume material** For the Subagent highlight, include why it is needed, exact technology, function, optimization, problem solved, measurable evidence, and non-claims. Do not claim token savings without a benchmark. -- [ ] **Step 4: Bump and validate release contract** +- [x] **Step 4: Bump and validate release contract** Set package version to `0.15.0a0`, lock it, update all exact version tests, and import stable Subagent API from installed-package smoke. @@ -930,7 +930,7 @@ py -m uv run pyright git diff --check ``` -- [ ] **Step 5: Commit documentation and release preparation** +- [x] **Step 5: Commit documentation and release preparation** ```powershell git add docs README.md SECURITY.md CHANGELOG.md pyproject.toml uv.lock tests diff --git a/pyproject.toml b/pyproject.toml index 724736e..5591486 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "mini-code-agent" -version = "0.14.0a0" +version = "0.15.0a0" description = "A framework-light, provider-neutral, enterprise-grade mini code agent." readme = "README.md" requires-python = ">=3.12,<3.14" diff --git a/src/mini_code_agent/policy/models.py b/src/mini_code_agent/policy/models.py index a4e5efe..ee7cf8c 100644 --- a/src/mini_code_agent/policy/models.py +++ b/src/mini_code_agent/policy/models.py @@ -34,6 +34,7 @@ class TrustSource(StrEnum): PROJECT = "project" MODEL = "model" EXTENSION = "extension" + SUBAGENT = "subagent" class PolicyRequest(BaseModel): diff --git a/src/mini_code_agent/subagents/__init__.py b/src/mini_code_agent/subagents/__init__.py new file mode 100644 index 0000000..1b9687d --- /dev/null +++ b/src/mini_code_agent/subagents/__init__.py @@ -0,0 +1,55 @@ +from mini_code_agent.subagents.contracts import ( + SubagentCompositionError, + SubagentProviderFactory, + SubagentToolFactory, +) +from mini_code_agent.subagents.events import ( + NullSubagentEventSink, + RecordingSubagentEventSink, + SubagentBatchCompleted, + SubagentBatchStarted, + SubagentCompleted, + SubagentEvent, + SubagentEventSink, + SubagentStarted, +) +from mini_code_agent.subagents.models import ( + SubagentBatchResult, + SubagentChildResult, + SubagentError, + SubagentErrorCode, + SubagentEvidenceItem, + SubagentLimits, + SubagentProfile, + SubagentStatus, +) +from mini_code_agent.subagents.supervisor import SubagentSupervisor +from mini_code_agent.subagents.tools import ( + SubagentAnalysisTool, + build_subagent_tools, +) + +__all__ = [ + "NullSubagentEventSink", + "RecordingSubagentEventSink", + "SubagentAnalysisTool", + "SubagentBatchCompleted", + "SubagentBatchResult", + "SubagentBatchStarted", + "SubagentChildResult", + "SubagentCompleted", + "SubagentCompositionError", + "SubagentError", + "SubagentErrorCode", + "SubagentEvent", + "SubagentEventSink", + "SubagentEvidenceItem", + "SubagentLimits", + "SubagentProfile", + "SubagentProviderFactory", + "SubagentStarted", + "SubagentStatus", + "SubagentSupervisor", + "SubagentToolFactory", + "build_subagent_tools", +] diff --git a/src/mini_code_agent/subagents/contracts.py b/src/mini_code_agent/subagents/contracts.py new file mode 100644 index 0000000..8fe0a4d --- /dev/null +++ b/src/mini_code_agent/subagents/contracts.py @@ -0,0 +1,61 @@ +from __future__ import annotations + +from pathlib import Path +from typing import Literal, Protocol + +from mini_code_agent.policy.models import TrustSource +from mini_code_agent.providers.base import ModelProvider +from mini_code_agent.subagents.models import SubagentProfile +from mini_code_agent.tools.base import SideEffect, ToolExecutor + + +class SubagentProviderFactory(Protocol): + def create( + self, + profile: SubagentProfile, + child_id: str, + ) -> ModelProvider: ... + + +class SubagentToolFactory(Protocol): + def create( + self, + profile: SubagentProfile, + workspace_root: Path, + ) -> ToolExecutor: ... + + +class GovernedSubagentTools(ToolExecutor, Protocol): + @property + def governance_enforced(self) -> Literal[True]: ... + + def trust_source_for(self, tool_name: str) -> TrustSource: ... + + +class SubagentCompositionError(RuntimeError): + def __init__(self) -> None: + super().__init__("Subagent capabilities did not match the host profile.") + + +def validate_child_tools( + profile: SubagentProfile, + tools: ToolExecutor, +) -> None: + try: + definitions = tools.definitions + names = tuple(definition.name for definition in definitions) + if names != profile.tool_names: + raise SubagentCompositionError + if any(definition.side_effect is not SideEffect.READ_ONLY for definition in definitions): + raise SubagentCompositionError + if getattr(tools, "governance_enforced", None) is not True: + raise SubagentCompositionError + trust_source_for = getattr(tools, "trust_source_for", None) + if not callable(trust_source_for): + raise SubagentCompositionError + if any(trust_source_for(name) is not TrustSource.SUBAGENT for name in names): + raise SubagentCompositionError + except SubagentCompositionError: + raise + except Exception: + raise SubagentCompositionError from None diff --git a/src/mini_code_agent/subagents/events.py b/src/mini_code_agent/subagents/events.py new file mode 100644 index 0000000..90ceb79 --- /dev/null +++ b/src/mini_code_agent/subagents/events.py @@ -0,0 +1,80 @@ +from __future__ import annotations + +from datetime import UTC, datetime +from typing import Literal, Protocol +from uuid import uuid4 + +from pydantic import BaseModel, ConfigDict, Field + +from mini_code_agent.providers.base import TokenUsage +from mini_code_agent.subagents.models import SubagentStatus + +_IDENTIFIER = r"^[A-Za-z0-9][A-Za-z0-9._-]{0,95}$" +_PROFILE_ID = r"^[a-z0-9][a-z0-9_-]{0,63}$" +_SHA256 = r"^[0-9a-f]{64}$" + + +def _event_id() -> str: + return str(uuid4()) + + +class SubagentEventBase(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + event_id: str = Field(default_factory=_event_id, pattern=_IDENTIFIER) + timestamp: datetime = Field(default_factory=lambda: datetime.now(UTC)) + parent_tool_call_id: str = Field(min_length=1, max_length=128) + profile_id: str = Field(pattern=_PROFILE_ID) + + +class SubagentBatchStarted(SubagentEventBase): + type: Literal["subagent_batch_started"] = "subagent_batch_started" + task_count: int = Field(ge=1, le=4) + + +class SubagentStarted(SubagentEventBase): + type: Literal["subagent_started"] = "subagent_started" + child_id: str = Field(pattern=_IDENTIFIER) + ordinal: int = Field(ge=0, le=3) + + +class SubagentCompleted(SubagentEventBase): + type: Literal["subagent_completed"] = "subagent_completed" + child_id: str = Field(pattern=_IDENTIFIER) + ordinal: int = Field(ge=0, le=3) + status: SubagentStatus + duration_ms: int = Field(ge=0, le=3_700_000) + turns: int = Field(ge=0, le=32) + tool_calls: int = Field(ge=0, le=128) + usage: TokenUsage = Field(default_factory=TokenUsage) + result_sha256: str = Field(pattern=_SHA256) + + +class SubagentBatchCompleted(SubagentEventBase): + type: Literal["subagent_batch_completed"] = "subagent_batch_completed" + duration_ms: int = Field(ge=0, le=3_700_000) + completed: int = Field(ge=0, le=4) + stopped: int = Field(ge=0, le=4) + timed_out: int = Field(ge=0, le=4) + failed: int = Field(ge=0, le=4) + result_sha256: str = Field(pattern=_SHA256) + + +SubagentEvent = SubagentBatchStarted | SubagentStarted | SubagentCompleted | SubagentBatchCompleted + + +class SubagentEventSink(Protocol): + def publish(self, event: SubagentEvent) -> None: ... + + +class NullSubagentEventSink: + def publish(self, event: SubagentEvent) -> None: + del event + + +class RecordingSubagentEventSink: + def __init__(self) -> None: + self.events: list[SubagentEvent] = [] + + def publish(self, event: SubagentEvent) -> None: + self.events.append(event) diff --git a/src/mini_code_agent/subagents/evidence.py b/src/mini_code_agent/subagents/evidence.py new file mode 100644 index 0000000..491db8a --- /dev/null +++ b/src/mini_code_agent/subagents/evidence.py @@ -0,0 +1,73 @@ +from __future__ import annotations + +import hashlib +import json + +from pydantic import BaseModel + +from mini_code_agent.agent.models import AgentResult +from mini_code_agent.domain.content import ToolCall, ToolResult +from mini_code_agent.subagents.models import SubagentEvidenceItem + + +class SubagentEvidenceError(RuntimeError): + def __init__(self) -> None: + super().__init__("Subagent transcript evidence was invalid.") + + +def extract_subagent_evidence( + result: AgentResult, + *, + max_items: int, +) -> tuple[SubagentEvidenceItem, ...]: + if not 0 <= max_items <= 256: + raise SubagentEvidenceError + + ordered_calls: list[ToolCall] = [] + calls: dict[str, ToolCall] = {} + results: dict[str, ToolResult] = {} + for message in result.messages: + for block in message.content: + if isinstance(block, ToolCall): + if block.id in calls or block.id in results: + raise SubagentEvidenceError + calls[block.id] = block + ordered_calls.append(block) + elif isinstance(block, ToolResult): + if block.tool_call_id not in calls or block.tool_call_id in results: + raise SubagentEvidenceError + results[block.tool_call_id] = block + + if ( + len(ordered_calls) != result.tool_calls + or len(ordered_calls) > max_items + or len(results) != len(ordered_calls) + ): + raise SubagentEvidenceError + + return tuple(_evidence_item(call, results[call.id]) for call in ordered_calls) + + +def subagent_result_sha256(value: BaseModel) -> str: + encoded = json.dumps( + value.model_dump(mode="json"), + ensure_ascii=True, + allow_nan=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + return hashlib.sha256(encoded).hexdigest() + + +def _evidence_item( + call: ToolCall, + result: ToolResult, +) -> SubagentEvidenceItem: + content = result.content.encode("utf-8") + return SubagentEvidenceItem( + tool_call_id=call.id, + tool_name=call.name, + is_error=result.is_error, + content_chars=len(result.content), + content_sha256=hashlib.sha256(content).hexdigest(), + ) diff --git a/src/mini_code_agent/subagents/models.py b/src/mini_code_agent/subagents/models.py new file mode 100644 index 0000000..3198016 --- /dev/null +++ b/src/mini_code_agent/subagents/models.py @@ -0,0 +1,252 @@ +from __future__ import annotations + +import hashlib +import json +from enum import StrEnum +from typing import Annotated, Literal, Self + +from pydantic import ( + BaseModel, + ConfigDict, + Field, + field_validator, + model_validator, +) + +from mini_code_agent.agent.models import AgentLimits, StopReason +from mini_code_agent.providers.base import TokenUsage + +_IDENTIFIER = r"^[A-Za-z0-9][A-Za-z0-9._-]{0,95}$" +_PROFILE_ID = r"^[a-z0-9][a-z0-9_-]{0,63}$" +_TOOL_NAME = r"^[a-z][a-z0-9_]{0,63}$" +_SHA256 = r"^[0-9a-f]{64}$" + +ToolName = Annotated[str, Field(pattern=_TOOL_NAME)] + + +class SubagentStatus(StrEnum): + COMPLETED = "completed" + STOPPED = "stopped" + TIMED_OUT = "timed_out" + FAILED = "failed" + BATCH_TIMED_OUT = "batch_timed_out" + + +class SubagentErrorCode(StrEnum): + INVALID_BATCH = "invalid_batch" + COMPOSITION_FAILED = "composition_failed" + CHILD_TIMEOUT = "child_timeout" + CHILD_FAILED = "child_failed" + BATCH_TIMEOUT = "batch_timeout" + RESULT_TOO_LARGE = "result_too_large" + + +class SubagentError(RuntimeError): + def __init__(self, code: SubagentErrorCode, public_message: str) -> None: + super().__init__(public_message) + self.code = code + self.public_message = public_message + + +class SubagentLimits(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + max_tasks: int = Field(default=4, ge=1, le=4) + max_concurrency: int = Field(default=2, ge=1, le=4) + max_task_chars: int = Field(default=4_000, ge=1, le=20_000) + child_timeout_seconds: float = Field(default=120.0, gt=0, le=600) + batch_timeout_seconds: float = Field(default=300.0, gt=0, le=900) + max_summary_chars: int = Field(default=8_000, ge=1, le=32_000) + max_evidence_items: int = Field(default=64, ge=0, le=256) + max_result_bytes: int = Field(default=131_072, ge=1, le=1_048_576) + + @model_validator(mode="after") + def validate_relationships(self) -> Self: + if self.max_concurrency > self.max_tasks: + raise ValueError("Subagent concurrency cannot exceed task count.") + if self.batch_timeout_seconds < self.child_timeout_seconds: + raise ValueError("Batch timeout cannot be lower than child timeout.") + return self + + +class SubagentProfile(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + profile_id: str = Field(pattern=_PROFILE_ID) + local_name: str = Field(pattern=_TOOL_NAME) + description: str = Field(min_length=1, max_length=500) + system_prompt: str = Field(min_length=1, max_length=20_000) + tool_names: tuple[ToolName, ...] = Field(min_length=1, max_length=16) + mode: Literal["analysis"] = "analysis" + agent_limits: AgentLimits = Field(default_factory=AgentLimits) + limits: SubagentLimits = Field(default_factory=SubagentLimits) + + @field_validator("description", "system_prompt") + @classmethod + def reject_nul_text(cls, value: str) -> str: + if "\0" in value: + raise ValueError("Subagent profile text cannot contain NUL.") + return value + + @model_validator(mode="after") + def validate_capabilities(self) -> Self: + if len(set(self.tool_names)) != len(self.tool_names): + raise ValueError("Subagent Tool names must be unique.") + if self.local_name in self.tool_names or any( + name.startswith("delegate_") for name in self.tool_names + ): + raise ValueError("Subagent profiles cannot recurse.") + if self.agent_limits.max_turns > 32 or self.agent_limits.max_tool_calls > 128: + raise ValueError("Subagent Agent limits exceed the hard ceiling.") + if self.agent_limits.max_tool_calls > self.limits.max_evidence_items: + raise ValueError("Every child ToolCall requires an evidence slot.") + return self + + +class SubagentEvidenceItem(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + tool_call_id: str = Field(min_length=1, max_length=128) + tool_name: str = Field(pattern=_TOOL_NAME) + is_error: bool + content_chars: int = Field(ge=1, le=16_777_216) + content_sha256: str = Field(pattern=_SHA256) + + +class SubagentChildResult(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + child_id: str = Field(pattern=_IDENTIFIER) + ordinal: int = Field(ge=0, le=3) + profile_id: str = Field(pattern=_PROFILE_ID) + status: SubagentStatus + stop_reason: StopReason | None = None + turns: int = Field(ge=0, le=32) + tool_calls: int = Field(ge=0, le=128) + usage: TokenUsage = Field(default_factory=TokenUsage) + untrusted_summary: str | None = Field(default=None, max_length=32_000) + evidence: tuple[SubagentEvidenceItem, ...] = Field(default=(), max_length=256) + error_code: SubagentErrorCode | None = None + error_message: str | None = Field(default=None, min_length=1, max_length=500) + result_sha256: str = Field(pattern=_SHA256) + + @field_validator("untrusted_summary") + @classmethod + def reject_nul_summary(cls, value: str | None) -> str | None: + if value is not None and "\0" in value: + raise ValueError("Subagent summary cannot contain NUL.") + return value + + @model_validator(mode="after") + def validate_status_projection(self) -> Self: + has_error = self.error_code is not None or self.error_message is not None + if (self.error_code is None) != (self.error_message is None): + raise ValueError("Subagent error code and message must be paired.") + if self.status is SubagentStatus.COMPLETED: + if self.stop_reason is not StopReason.COMPLETED or has_error: + raise ValueError("Completed Subagent result is inconsistent.") + elif self.status is SubagentStatus.STOPPED: + if self.stop_reason is None or self.stop_reason is StopReason.COMPLETED or has_error: + raise ValueError("Stopped Subagent result is inconsistent.") + else: + expected = { + SubagentStatus.TIMED_OUT: SubagentErrorCode.CHILD_TIMEOUT, + SubagentStatus.FAILED: SubagentErrorCode.CHILD_FAILED, + SubagentStatus.BATCH_TIMED_OUT: SubagentErrorCode.BATCH_TIMEOUT, + }[self.status] + if ( + self.stop_reason is not None + or self.error_code is not expected + or self.error_message is None + or self.turns != 0 + or self.tool_calls != 0 + or self.untrusted_summary is not None + or self.evidence + ): + raise ValueError("Failed Subagent result is inconsistent.") + return self + + +class SubagentBatchResult(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + profile_id: str = Field(pattern=_PROFILE_ID) + children: tuple[SubagentChildResult, ...] = Field(min_length=1, max_length=4) + duration_ms: int = Field(ge=0, le=3_700_000) + completed: int = Field(ge=0, le=4) + stopped: int = Field(ge=0, le=4) + timed_out: int = Field(ge=0, le=4) + failed: int = Field(ge=0, le=4) + result_sha256: str = Field(pattern=_SHA256) + + @classmethod + def from_children( + cls, + *, + profile_id: str, + children: tuple[SubagentChildResult, ...], + duration_ms: int, + ) -> Self: + counts = _status_counts(children) + projection: dict[str, object] = { + "profile_id": profile_id, + "children": [child.model_dump(mode="json") for child in children], + "duration_ms": duration_ms, + **counts, + } + return cls( + profile_id=profile_id, + children=children, + duration_ms=duration_ms, + completed=counts["completed"], + stopped=counts["stopped"], + timed_out=counts["timed_out"], + failed=counts["failed"], + result_sha256=_canonical_sha256(projection), + ) + + @model_validator(mode="after") + def validate_projection(self) -> Self: + if tuple(child.ordinal for child in self.children) != tuple(range(len(self.children))): + raise ValueError("Subagent child ordinals must be contiguous.") + if len({child.child_id for child in self.children}) != len(self.children): + raise ValueError("Subagent child identifiers must be unique.") + if any(child.profile_id != self.profile_id for child in self.children): + raise ValueError("Subagent child profile does not match the batch.") + counts = _status_counts(self.children) + if any(getattr(self, key) != value for key, value in counts.items()): + raise ValueError("Subagent batch counts are inconsistent.") + projection = { + "profile_id": self.profile_id, + "children": [child.model_dump(mode="json") for child in self.children], + "duration_ms": self.duration_ms, + **counts, + } + if self.result_sha256 != _canonical_sha256(projection): + raise ValueError("Subagent batch hash is inconsistent.") + return self + + +def _status_counts( + children: tuple[SubagentChildResult, ...], +) -> dict[str, int]: + return { + "completed": sum(child.status is SubagentStatus.COMPLETED for child in children), + "stopped": sum(child.status is SubagentStatus.STOPPED for child in children), + "timed_out": sum( + child.status in {SubagentStatus.TIMED_OUT, SubagentStatus.BATCH_TIMED_OUT} + for child in children + ), + "failed": sum(child.status is SubagentStatus.FAILED for child in children), + } + + +def _canonical_sha256(value: object) -> str: + encoded = json.dumps( + value, + ensure_ascii=True, + allow_nan=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + return hashlib.sha256(encoded).hexdigest() diff --git a/src/mini_code_agent/subagents/supervisor.py b/src/mini_code_agent/subagents/supervisor.py new file mode 100644 index 0000000..5e8045c --- /dev/null +++ b/src/mini_code_agent/subagents/supervisor.py @@ -0,0 +1,420 @@ +from __future__ import annotations + +import asyncio +import hashlib +import json +import re +import time +from collections.abc import Callable +from dataclasses import dataclass +from pathlib import Path +from typing import cast +from uuid import uuid4 + +from mini_code_agent.agent.models import AgentResult, StopReason +from mini_code_agent.agent.runtime import AgentRuntime +from mini_code_agent.providers.base import ProviderCapabilities, TokenUsage +from mini_code_agent.subagents.contracts import ( + SubagentCompositionError, + SubagentProviderFactory, + SubagentToolFactory, + validate_child_tools, +) +from mini_code_agent.subagents.events import ( + NullSubagentEventSink, + SubagentBatchCompleted, + SubagentBatchStarted, + SubagentCompleted, + SubagentEvent, + SubagentEventSink, + SubagentStarted, +) +from mini_code_agent.subagents.evidence import ( + SubagentEvidenceError, + extract_subagent_evidence, +) +from mini_code_agent.subagents.models import ( + SubagentBatchResult, + SubagentChildResult, + SubagentError, + SubagentErrorCode, + SubagentProfile, + SubagentStatus, +) + +_CHILD_ID = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]{0,95}$") +_INVALID_BATCH_MESSAGE = "Subagent batch request was invalid." +_CHILD_TIMEOUT_MESSAGE = "Subagent execution timed out." +_CHILD_FAILED_MESSAGE = "Subagent execution failed." + + +@dataclass(frozen=True, slots=True) +class _PreparedChild: + child_id: str + ordinal: int + task: str + runtime: AgentRuntime + + +class SubagentSupervisor: + def __init__( + self, + profile: SubagentProfile, + *, + workspace_root: Path, + provider_factory: SubagentProviderFactory, + tool_factory: SubagentToolFactory, + events: SubagentEventSink | None = None, + id_factory: Callable[[], str] | None = None, + monotonic: Callable[[], float] = time.monotonic, + ) -> None: + try: + root = workspace_root.resolve(strict=True) + except OSError: + raise ValueError("Subagent workspace root must exist.") from None + if not root.is_dir(): + raise ValueError("Subagent workspace root must be a directory.") + self._profile = profile + self._workspace_root = root + self._provider_factory = provider_factory + self._tool_factory = tool_factory + self._events = events or NullSubagentEventSink() + self._id_factory = id_factory or (lambda: str(uuid4())) + self._monotonic = monotonic + + @property + def profile(self) -> SubagentProfile: + return self._profile + + async def run_batch( + self, + *, + parent_tool_call_id: str, + tasks: tuple[str, ...], + ) -> SubagentBatchResult: + self._validate_batch(parent_tool_call_id, tasks) + children = self._prepare_children(tasks) + started_at = self._monotonic() + self._publish( + SubagentBatchStarted( + parent_tool_call_id=parent_tool_call_id, + profile_id=self._profile.profile_id, + task_count=len(tasks), + ) + ) + result_slots: list[SubagentChildResult | None] = [None] * len(children) + semaphore = asyncio.Semaphore(self._profile.limits.max_concurrency) + + async def run_at(child: _PreparedChild) -> None: + result_slots[child.ordinal] = await self._run_child( + parent_tool_call_id=parent_tool_call_id, + child=child, + semaphore=semaphore, + ) + + try: + async with asyncio.timeout(self._profile.limits.batch_timeout_seconds): + async with asyncio.TaskGroup() as group: + for child in children: + group.create_task(run_at(child)) + except TimeoutError: + duration_ms = _elapsed_ms(started_at, self._monotonic()) + for child in children: + if result_slots[child.ordinal] is not None: + continue + result = self._error_result( + child, + status=SubagentStatus.BATCH_TIMED_OUT, + code=SubagentErrorCode.BATCH_TIMEOUT, + message="Subagent batch timed out.", + ) + result_slots[child.ordinal] = result + self._publish_child_completed( + parent_tool_call_id=parent_tool_call_id, + result=result, + duration_ms=duration_ms, + ) + + duration_ms = _elapsed_ms(started_at, self._monotonic()) + if any(result is None for result in result_slots): + raise RuntimeError("Subagent result slot was not populated.") + results = cast(tuple[SubagentChildResult, ...], tuple(result_slots)) + batch = SubagentBatchResult.from_children( + profile_id=self._profile.profile_id, + children=results, + duration_ms=duration_ms, + ) + self._publish( + SubagentBatchCompleted( + parent_tool_call_id=parent_tool_call_id, + profile_id=self._profile.profile_id, + duration_ms=batch.duration_ms, + completed=batch.completed, + stopped=batch.stopped, + timed_out=batch.timed_out, + failed=batch.failed, + result_sha256=batch.result_sha256, + ) + ) + return batch + + def _prepare_children( + self, + tasks: tuple[str, ...], + ) -> tuple[_PreparedChild, ...]: + prepared: list[_PreparedChild] = [] + provider_ids: set[int] = set() + tool_ids: set[int] = set() + try: + child_ids = tuple(self._id_factory() for _ in tasks) + if len(set(child_ids)) != len(child_ids) or any( + _CHILD_ID.fullmatch(child_id) is None for child_id in child_ids + ): + raise SubagentCompositionError + for ordinal, (task, child_id) in enumerate(zip(tasks, child_ids, strict=True)): + provider = self._provider_factory.create(self._profile, child_id) + tools = self._tool_factory.create(self._profile, self._workspace_root) + if id(provider) in provider_ids or id(tools) in tool_ids: + raise SubagentCompositionError + _validate_provider(provider) + validate_child_tools(self._profile, tools) + provider_ids.add(id(provider)) + tool_ids.add(id(tools)) + prepared.append( + _PreparedChild( + child_id=child_id, + ordinal=ordinal, + task=task, + runtime=AgentRuntime( + provider, + tools, + limits=self._profile.agent_limits, + ), + ) + ) + except asyncio.CancelledError: + raise + except SubagentCompositionError: + raise + except Exception: + raise SubagentCompositionError from None + return tuple(prepared) + + async def _run_child( + self, + *, + parent_tool_call_id: str, + child: _PreparedChild, + semaphore: asyncio.Semaphore, + ) -> SubagentChildResult: + started_at = self._monotonic() + self._publish( + SubagentStarted( + parent_tool_call_id=parent_tool_call_id, + profile_id=self._profile.profile_id, + child_id=child.child_id, + ordinal=child.ordinal, + ) + ) + try: + async with semaphore: + async with asyncio.timeout(self._profile.limits.child_timeout_seconds): + candidate = cast( + object, + await child.runtime.run( + user_prompt=child.task, + system_prompt=self._profile.system_prompt, + run_id=_runtime_id(child.child_id), + ), + ) + if not isinstance(candidate, AgentResult): + raise TypeError("invalid Agent result") + result = self._project_agent_result(child, candidate) + except asyncio.CancelledError: + raise + except TimeoutError: + result = self._error_result( + child, + status=SubagentStatus.TIMED_OUT, + code=SubagentErrorCode.CHILD_TIMEOUT, + message=_CHILD_TIMEOUT_MESSAGE, + ) + except Exception: + result = self._error_result( + child, + status=SubagentStatus.FAILED, + code=SubagentErrorCode.CHILD_FAILED, + message=_CHILD_FAILED_MESSAGE, + ) + duration_ms = _elapsed_ms(started_at, self._monotonic()) + self._publish_child_completed( + parent_tool_call_id=parent_tool_call_id, + result=result, + duration_ms=duration_ms, + ) + return result + + def _publish_child_completed( + self, + *, + parent_tool_call_id: str, + result: SubagentChildResult, + duration_ms: int, + ) -> None: + self._publish( + SubagentCompleted( + parent_tool_call_id=parent_tool_call_id, + profile_id=self._profile.profile_id, + child_id=result.child_id, + ordinal=result.ordinal, + status=result.status, + duration_ms=duration_ms, + turns=result.turns, + tool_calls=result.tool_calls, + usage=result.usage, + result_sha256=result.result_sha256, + ) + ) + + def _project_agent_result( + self, + child: _PreparedChild, + result: AgentResult, + ) -> SubagentChildResult: + evidence = extract_subagent_evidence( + result, + max_items=self._profile.limits.max_evidence_items, + ) + if result.turns > self._profile.agent_limits.max_turns: + raise SubagentEvidenceError + if result.tool_calls > self._profile.agent_limits.max_tool_calls: + raise SubagentEvidenceError + status = ( + SubagentStatus.COMPLETED + if result.stop_reason is StopReason.COMPLETED + else SubagentStatus.STOPPED + ) + summary = result.final_text + if summary is not None: + summary = summary[: self._profile.limits.max_summary_chars] + if "\0" in summary: + raise SubagentEvidenceError + projection: dict[str, object] = { + "child_id": child.child_id, + "ordinal": child.ordinal, + "profile_id": self._profile.profile_id, + "status": status.value, + "stop_reason": result.stop_reason.value, + "turns": result.turns, + "tool_calls": result.tool_calls, + "usage": result.usage.model_dump(mode="json"), + "untrusted_summary": summary, + "evidence": [item.model_dump(mode="json") for item in evidence], + "error_code": None, + "error_message": None, + } + return SubagentChildResult.model_validate( + projection | {"result_sha256": _canonical_sha256(projection)} + ) + + def _error_result( + self, + child: _PreparedChild, + *, + status: SubagentStatus, + code: SubagentErrorCode, + message: str, + ) -> SubagentChildResult: + projection: dict[str, object] = { + "child_id": child.child_id, + "ordinal": child.ordinal, + "profile_id": self._profile.profile_id, + "status": status.value, + "stop_reason": None, + "turns": 0, + "tool_calls": 0, + "usage": TokenUsage().model_dump(mode="json"), + "untrusted_summary": None, + "evidence": [], + "error_code": code.value, + "error_message": message, + } + return SubagentChildResult.model_validate( + projection | {"result_sha256": _canonical_sha256(projection)} + ) + + def _validate_batch( + self, + parent_tool_call_id: str, + tasks: tuple[str, ...], + ) -> None: + limits = self._profile.limits + valid_parent = _valid_parent_tool_call_id(parent_tool_call_id) + valid_tasks = _valid_tasks( + tasks, + max_tasks=limits.max_tasks, + max_task_chars=limits.max_task_chars, + ) + if not valid_parent or not valid_tasks: + raise SubagentError( + SubagentErrorCode.INVALID_BATCH, + _INVALID_BATCH_MESSAGE, + ) + + def _publish(self, event: SubagentEvent) -> None: + try: + self._events.publish(event) + except Exception: + return + + +def _validate_provider(provider: object) -> None: + capabilities = getattr(provider, "capabilities", None) + if ( + not isinstance(capabilities, ProviderCapabilities) + or not callable(getattr(provider, "complete", None)) + or not callable(getattr(provider, "stream", None)) + ): + raise SubagentCompositionError + + +def _valid_parent_tool_call_id(value: object) -> bool: + return isinstance(value, str) and 1 <= len(value) <= 128 and "\0" not in value + + +def _valid_tasks( + value: object, + *, + max_tasks: int, + max_task_chars: int, +) -> bool: + if not isinstance(value, tuple): + return False + tasks = cast(tuple[object, ...], value) + if not 1 <= len(tasks) <= max_tasks or not all( + isinstance(task, str) and 1 <= len(task) <= max_task_chars and "\0" not in task + for task in tasks + ): + return False + string_tasks = cast(tuple[str, ...], tasks) + return len(set(string_tasks)) == len(string_tasks) + + +def _runtime_id(child_id: str) -> str: + digest = hashlib.sha256(child_id.encode("utf-8")).hexdigest()[:32] + return f"subagent-{digest}" + + +def _elapsed_ms(started_at: float, completed_at: float) -> int: + return max(0, int((completed_at - started_at) * 1000)) + + +def _canonical_sha256(value: object) -> str: + encoded = json.dumps( + value, + ensure_ascii=True, + allow_nan=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + return hashlib.sha256(encoded).hexdigest() diff --git a/src/mini_code_agent/subagents/tools.py b/src/mini_code_agent/subagents/tools.py new file mode 100644 index 0000000..a161ffe --- /dev/null +++ b/src/mini_code_agent/subagents/tools.py @@ -0,0 +1,268 @@ +from __future__ import annotations + +import asyncio +import json +from collections.abc import Iterable +from typing import Protocol + +from pydantic import ( + BaseModel, + ConfigDict, + Field, + JsonValue, + ValidationError, + field_validator, +) + +from mini_code_agent.domain.content import ToolCall, ToolResult +from mini_code_agent.policy.models import ActionPreview, RiskLevel +from mini_code_agent.subagents.contracts import SubagentCompositionError +from mini_code_agent.subagents.models import ( + SubagentBatchResult, + SubagentError, + SubagentErrorCode, + SubagentProfile, +) +from mini_code_agent.tools.base import SideEffect, ToolDefinition + +_INVALID_ARGUMENTS_MESSAGE = "Subagent Tool arguments were invalid." +_FAILED_MESSAGE = "Subagent batch execution failed." +_COMPOSITION_MESSAGE = "Subagent capabilities did not match the host profile." +_RESULT_TOO_LARGE_MESSAGE = "Subagent batch result exceeded the configured size limit." +_PROFILE_CONFLICT_MESSAGE = "Subagent Tool profiles conflict." + + +class _SubagentBatchRunner(Protocol): + @property + def profile(self) -> SubagentProfile: ... + + async def run_batch( + self, + *, + parent_tool_call_id: str, + tasks: tuple[str, ...], + ) -> SubagentBatchResult: ... + + +class _AnalysisArguments(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True, strict=True) + + tasks: tuple[str, ...] = Field(min_length=1, max_length=4) + reason: str = Field(min_length=1, max_length=500) + + @field_validator("tasks") + @classmethod + def reject_invalid_tasks(cls, value: tuple[str, ...]) -> tuple[str, ...]: + if len(set(value)) != len(value) or any("\0" in task for task in value): + raise ValueError("Subagent tasks were invalid.") + return value + + @field_validator("reason") + @classmethod + def reject_invalid_reason(cls, value: str) -> str: + if "\0" in value: + raise ValueError("Subagent reason was invalid.") + return value + + +class SubagentAnalysisTool: + def __init__(self, supervisor: _SubagentBatchRunner) -> None: + self._supervisor = supervisor + self._profile = supervisor.profile + self._definition = ToolDefinition( + name=self._profile.local_name, + description=self._profile.description, + input_schema=_input_schema(self._profile), + side_effect=SideEffect.READ_ONLY, + ) + + @property + def definition(self) -> ToolDefinition: + return self._definition + + async def preview(self, call: ToolCall) -> ActionPreview: + if call.name != self._definition.name: + raise ValueError("The requested Subagent Tool is not registered.") + arguments = self._parse_arguments(call) + return ActionPreview( + tool_call_id=call.id, + tool_name=self._definition.name, + side_effect=SideEffect.READ_ONLY, + risk=RiskLevel.MEDIUM, + summary=( + f"Delegate {len(arguments.tasks)} isolated analysis " + f"task(s) to profile {self._profile.profile_id}." + ), + reason=arguments.reason, + resources=(".",), + ) + + async def execute(self, call: ToolCall) -> ToolResult: + if call.name != self._definition.name: + return _error( + call.id, + "unknown_tool", + "The requested Subagent Tool is not registered.", + ) + try: + arguments = self._parse_arguments(call) + except ValueError: + return _error( + call.id, + "invalid_arguments", + _INVALID_ARGUMENTS_MESSAGE, + ) + + try: + candidate = await self._supervisor.run_batch( + parent_tool_call_id=call.id, + tasks=arguments.tasks, + ) + batch = _validated_batch( + candidate, + profile=self._profile, + expected_children=len(arguments.tasks), + ) + content = _serialize_batch( + batch, + max_result_bytes=self._profile.limits.max_result_bytes, + ) + except asyncio.CancelledError: + raise + except SubagentError as exc: + return _error(call.id, exc.code.value, exc.public_message) + except SubagentCompositionError: + return _error( + call.id, + SubagentErrorCode.COMPOSITION_FAILED.value, + _COMPOSITION_MESSAGE, + ) + except _ResultTooLarge: + return _error( + call.id, + SubagentErrorCode.RESULT_TOO_LARGE.value, + _RESULT_TOO_LARGE_MESSAGE, + ) + except Exception: + return _error( + call.id, + SubagentErrorCode.CHILD_FAILED.value, + _FAILED_MESSAGE, + ) + return ToolResult(tool_call_id=call.id, content=content) + + def _parse_arguments(self, call: ToolCall) -> _AnalysisArguments: + try: + arguments = _AnalysisArguments.model_validate( + dict(call.arguments), + strict=True, + ) + except ValidationError: + raise ValueError(_INVALID_ARGUMENTS_MESSAGE) from None + limits = self._profile.limits + if len(arguments.tasks) > limits.max_tasks or any( + len(task) > limits.max_task_chars for task in arguments.tasks + ): + raise ValueError(_INVALID_ARGUMENTS_MESSAGE) + return arguments + + +def build_subagent_tools( + supervisors: Iterable[_SubagentBatchRunner], +) -> tuple[SubagentAnalysisTool, ...]: + ordered = tuple(supervisors) + profiles = tuple(supervisor.profile for supervisor in ordered) + profile_ids = tuple(profile.profile_id for profile in profiles) + local_names = tuple(profile.local_name for profile in profiles) + child_tool_names = {tool_name for profile in profiles for tool_name in profile.tool_names} + if ( + len(set(profile_ids)) != len(profile_ids) + or len(set(local_names)) != len(local_names) + or not set(local_names).isdisjoint(child_tool_names) + ): + raise ValueError(_PROFILE_CONFLICT_MESSAGE) + return tuple(SubagentAnalysisTool(supervisor) for supervisor in ordered) + + +class _ResultTooLarge(ValueError): + pass + + +def _input_schema(profile: SubagentProfile) -> dict[str, JsonValue]: + no_nul = r"^[^\u0000]+$" + return { + "type": "object", + "properties": { + "tasks": { + "type": "array", + "minItems": 1, + "maxItems": profile.limits.max_tasks, + "uniqueItems": True, + "items": { + "type": "string", + "minLength": 1, + "maxLength": profile.limits.max_task_chars, + "pattern": no_nul, + }, + }, + "reason": { + "type": "string", + "minLength": 1, + "maxLength": 500, + "pattern": no_nul, + }, + }, + "required": ["tasks", "reason"], + "additionalProperties": False, + } + + +def _validated_batch( + candidate: object, + *, + profile: SubagentProfile, + expected_children: int, +) -> SubagentBatchResult: + if not isinstance(candidate, SubagentBatchResult): + raise ValueError("Invalid Subagent batch result.") + batch = SubagentBatchResult.model_validate(candidate.model_dump(mode="json")) + if batch.profile_id != profile.profile_id or len(batch.children) != expected_children: + raise ValueError("Invalid Subagent batch result.") + return batch + + +def _serialize_batch( + batch: SubagentBatchResult, + *, + max_result_bytes: int, +) -> str: + payload = { + "content_type": "subagent_batch_result", + **batch.model_dump(mode="json"), + } + try: + encoded = json.dumps( + payload, + ensure_ascii=True, + allow_nan=False, + separators=(",", ":"), + sort_keys=True, + ).encode("utf-8") + except (TypeError, ValueError, OverflowError): + raise ValueError("Invalid Subagent batch result.") from None + if len(encoded) > max_result_bytes: + raise _ResultTooLarge + return encoded.decode("ascii") + + +def _error(call_id: str, code: str, message: str) -> ToolResult: + return ToolResult( + tool_call_id=call_id, + content=json.dumps( + {"error": {"code": code, "message": message}}, + ensure_ascii=True, + separators=(",", ":"), + sort_keys=True, + ), + is_error=True, + ) diff --git a/tests/cli/test_cli.py b/tests/cli/test_cli.py index c8fd546..deb22e1 100644 --- a/tests/cli/test_cli.py +++ b/tests/cli/test_cli.py @@ -27,7 +27,7 @@ def test_version_option_prints_package_version() -> None: result = runner.invoke(app, ["--version"]) assert result.exit_code == 0 - assert result.stdout.strip() == "0.14.0a0" + assert result.stdout.strip() == "0.15.0a0" def test_module_entrypoint_prints_package_version() -> None: @@ -39,7 +39,7 @@ def test_module_entrypoint_prints_package_version() -> None: ) assert result.returncode == 0 - assert result.stdout.strip() == "0.14.0a0" + assert result.stdout.strip() == "0.15.0a0" def test_doctor_json_never_prints_secrets( diff --git a/tests/integration/test_agent_loop.py b/tests/integration/test_agent_loop.py index c4a728b..d309e77 100644 --- a/tests/integration/test_agent_loop.py +++ b/tests/integration/test_agent_loop.py @@ -53,7 +53,7 @@ async def test_fake_provider_drives_native_tool_call_round_trip() -> None: assert tool_result_message.role is MessageRole.USER assert tool_result_message.tool_results[0].tool_call_id == "call-1" payload = json.loads(tool_result_message.tool_results[0].content) - assert payload["package_version"] == "0.14.0a0" + assert payload["package_version"] == "0.15.0a0" assert [type(event) for event in events.events] == [ RunStarted, ModelStarted, diff --git a/tests/integration/test_governed_subagent_agent.py b/tests/integration/test_governed_subagent_agent.py new file mode 100644 index 0000000..fe7380d --- /dev/null +++ b/tests/integration/test_governed_subagent_agent.py @@ -0,0 +1,522 @@ +from __future__ import annotations + +import asyncio +import json +from collections import deque +from collections.abc import AsyncIterator +from pathlib import Path +from typing import cast + +import pytest + +from mini_code_agent.agent.models import AgentLimits, StopReason +from mini_code_agent.agent.runtime import AgentRuntime +from mini_code_agent.domain.content import ToolCall +from mini_code_agent.domain.messages import Message, MessageRole +from mini_code_agent.policy.approval import StaticApprovalHandler +from mini_code_agent.policy.engine import PolicyEngine +from mini_code_agent.policy.executor import GovernedToolExecutor +from mini_code_agent.policy.models import ( + PolicyDecision, + PolicyRule, + SessionMode, + TrustSource, +) +from mini_code_agent.providers.base import ( + FinishReason, + ModelProvider, + ModelRequest, + ModelResponse, + ProviderCapabilities, + ProviderStreamEvent, + ResponseCompleted, +) +from mini_code_agent.providers.fake import ScriptedProvider +from mini_code_agent.subagents.events import RecordingSubagentEventSink +from mini_code_agent.subagents.models import ( + SubagentLimits, + SubagentProfile, +) +from mini_code_agent.subagents.supervisor import SubagentSupervisor +from mini_code_agent.subagents.tools import build_subagent_tools +from mini_code_agent.tools.base import ToolExecutor +from mini_code_agent.tools.read_file import ReadFileTool +from mini_code_agent.tools.registry import ToolRegistry +from mini_code_agent.tools.search_text import SearchTextTool +from mini_code_agent.workspace.boundary import WorkspaceBoundary + + +def profile_for( + *, + child_timeout_seconds: float = 1, + batch_timeout_seconds: float = 3, +) -> SubagentProfile: + return SubagentProfile( + profile_id="review", + local_name="delegate_analysis", + description="Delegate isolated read-only code analysis.", + system_prompt="Use only the assigned read-only tools and return a brief summary.", + tool_names=("read_file", "search_text"), + agent_limits=AgentLimits( + max_turns=4, + max_tool_calls=4, + provider_timeout_seconds=1, + tool_timeout_seconds=1, + ), + limits=SubagentLimits( + max_tasks=4, + max_concurrency=2, + max_task_chars=1_000, + child_timeout_seconds=child_timeout_seconds, + batch_timeout_seconds=batch_timeout_seconds, + max_summary_chars=1_000, + max_evidence_items=4, + max_result_bytes=64_000, + ), + ) + + +def tool_response(call: ToolCall) -> ModelResponse: + return ModelResponse( + message=Message( + role=MessageRole.ASSISTANT, + content=(call,), + ), + finish_reason=FinishReason.TOOL_CALL, + ) + + +def stop_response(text: str) -> ModelResponse: + return ModelResponse( + message=Message.assistant_text(text), + finish_reason=FinishReason.STOP, + ) + + +def child_read_provider(summary: str = "Read review complete.") -> ScriptedProvider: + return ScriptedProvider( + ( + tool_response( + ToolCall( + id="read-1", + name="read_file", + arguments={"path": "src/app.py"}, + ) + ), + stop_response(summary), + ) + ) + + +def child_search_provider( + summary: str = "Search review complete.", +) -> ScriptedProvider: + return ScriptedProvider( + ( + tool_response( + ToolCall( + id="search-1", + name="search_text", + arguments={"query": "needle", "glob": "*.py"}, + ) + ), + stop_response(summary), + ) + ) + + +def parent_provider_for( + tasks: tuple[str, ...], + *, + final_text: str = "Delegation complete.", +) -> ScriptedProvider: + return ScriptedProvider( + ( + tool_response( + ToolCall( + id="delegate-1", + name="delegate_analysis", + arguments={ + "tasks": list(tasks), + "reason": "Independent bounded review.", + }, + ) + ), + stop_response(final_text), + ) + ) + + +class RecordingProviderFactory: + def __init__(self, providers: tuple[ModelProvider, ...]) -> None: + self._providers = deque(providers) + self.calls: list[tuple[str, str]] = [] + + def create( + self, + profile: SubagentProfile, + child_id: str, + ) -> ModelProvider: + self.calls.append((profile.profile_id, child_id)) + return self._providers.popleft() + + +class RealReadOnlyToolFactory: + def __init__(self) -> None: + self.executors: list[GovernedToolExecutor] = [] + + def create( + self, + profile: SubagentProfile, + workspace_root: Path, + ) -> ToolExecutor: + workspace = WorkspaceBoundary(workspace_root) + registry = ToolRegistry( + ( + ReadFileTool(workspace), + SearchTextTool(workspace), + ) + ) + executor = GovernedToolExecutor( + registry, + policy=PolicyEngine(), + approval=StaticApprovalHandler(approved=False), + session_mode=SessionMode.NON_INTERACTIVE, + trust_source=TrustSource.SUBAGENT, + ) + assert tuple(item.name for item in executor.definitions) == profile.tool_names + self.executors.append(executor) + return executor + + +def parent_executor( + supervisor: SubagentSupervisor, + *, + policy: PolicyEngine | None = None, +) -> GovernedToolExecutor: + return GovernedToolExecutor( + ToolRegistry(build_subagent_tools((supervisor,))), + policy=policy or PolicyEngine(), + approval=StaticApprovalHandler(approved=False), + session_mode=SessionMode.NON_INTERACTIVE, + trust_source=TrustSource.MODEL, + ) + + +def supervisor_for( + tmp_path: Path, + *, + profile: SubagentProfile, + providers: tuple[ModelProvider, ...], + events: RecordingSubagentEventSink | None = None, +) -> tuple[ + SubagentSupervisor, + RecordingProviderFactory, + RealReadOnlyToolFactory, +]: + provider_factory = RecordingProviderFactory(providers) + tool_factory = RealReadOnlyToolFactory() + child_ids = iter(f"child-{index + 1}" for index in range(len(providers))) + supervisor = SubagentSupervisor( + profile, + workspace_root=tmp_path, + provider_factory=provider_factory, + tool_factory=tool_factory, + events=events, + id_factory=lambda: next(child_ids), + ) + return supervisor, provider_factory, tool_factory + + +def workspace_snapshot(root: Path) -> dict[str, bytes]: + return { + path.relative_to(root).as_posix(): path.read_bytes() + for path in sorted(root.rglob("*")) + if path.is_file() + } + + +def prepare_workspace(tmp_path: Path) -> dict[str, bytes]: + source = tmp_path / "src" / "app.py" + source.parent.mkdir() + source.write_bytes(b"def run():\n return 'needle'\n") + return workspace_snapshot(tmp_path) + + +def delegated_payload(parent: ScriptedProvider) -> dict[str, object]: + result_message = parent.requests[1].messages[-1] + result = result_message.tool_results[0] + return cast(dict[str, object], json.loads(result.content)) + + +@pytest.mark.asyncio +async def test_real_parent_and_children_run_governed_read_only_path( + tmp_path: Path, +) -> None: + before = prepare_workspace(tmp_path) + tasks = ("Inspect parser behavior.", "Find needle references.") + children = (child_read_provider(), child_search_provider()) + events = RecordingSubagentEventSink() + supervisor, provider_factory, tool_factory = supervisor_for( + tmp_path, + profile=profile_for(), + providers=children, + events=events, + ) + parent = parent_provider_for(tasks) + + result = await AgentRuntime( + parent, + parent_executor(supervisor), + ).run( + user_prompt="Delegate two independent reviews.", + run_id="parent-subagent-run", + ) + + assert result.stop_reason is StopReason.COMPLETED + assert result.tool_calls == 1 + assert provider_factory.calls == [ + ("review", "child-1"), + ("review", "child-2"), + ] + assert children[0].requests[0].messages == (Message.user_text(tasks[0]),) + assert children[1].requests[0].messages == (Message.user_text(tasks[1]),) + child_run_ids = { + request.request_id.rsplit(":", 1)[0] for child in children for request in child.requests[:1] + } + assert len(child_run_ids) == 2 + assert all(run_id.startswith("subagent-") for run_id in child_run_ids) + + payload = delegated_payload(parent) + assert payload["content_type"] == "subagent_batch_result" + projected_children = cast(list[dict[str, object]], payload["children"]) + assert [child["ordinal"] for child in projected_children] == [0, 1] + assert [child["untrusted_summary"] for child in projected_children] == [ + "Read review complete.", + "Search review complete.", + ] + evidence = [cast(list[dict[str, object]], child["evidence"])[0] for child in projected_children] + assert [item["tool_name"] for item in evidence] == [ + "read_file", + "search_text", + ] + assert all(len(cast(str, item["content_sha256"])) == 64 for item in evidence) + assert all( + executor.trust_source_for(name) is TrustSource.SUBAGENT + for executor in tool_factory.executors + for name in ("read_file", "search_text") + ) + + event_payload = json.dumps( + [event.model_dump(mode="json") for event in events.events], + ensure_ascii=True, + sort_keys=True, + ) + for secret in ( + *tasks, + supervisor.profile.system_prompt, + "return 'needle'", + ): + assert secret not in event_payload + assert workspace_snapshot(tmp_path) == before + + +@pytest.mark.asyncio +async def test_parent_policy_deny_prevents_child_composition( + tmp_path: Path, +) -> None: + prepare_workspace(tmp_path) + supervisor, provider_factory, tool_factory = supervisor_for( + tmp_path, + profile=profile_for(), + providers=(child_read_provider(),), + ) + parent = parent_provider_for(("Inspect parser behavior.",)) + policy = PolicyEngine( + ( + PolicyRule( + id="deny-delegation", + decision=PolicyDecision.DENY, + rationale="Delegation disabled.", + tool_glob="delegate_analysis", + ), + ) + ) + + result = await AgentRuntime( + parent, + parent_executor(supervisor, policy=policy), + ).run( + user_prompt="Delegate review.", + run_id="denied-subagent-run", + ) + + assert result.stop_reason is StopReason.COMPLETED + denied = parent.requests[1].messages[-1].tool_results[0] + assert json.loads(denied.content)["error"]["code"] == "permission_denied" + assert provider_factory.calls == [] + assert tool_factory.executors == [] + + +@pytest.mark.asyncio +async def test_child_cannot_recursively_call_parent_delegation_tool( + tmp_path: Path, +) -> None: + prepare_workspace(tmp_path) + recursive = ScriptedProvider( + ( + tool_response( + ToolCall( + id="recursive-1", + name="delegate_analysis", + arguments={ + "tasks": ["nested"], + "reason": "Try recursion.", + }, + ) + ), + stop_response("Recursion was unavailable."), + ) + ) + supervisor, _, _ = supervisor_for( + tmp_path, + profile=profile_for(), + providers=(recursive,), + ) + parent = parent_provider_for(("Inspect recursion boundary.",)) + + result = await AgentRuntime( + parent, + parent_executor(supervisor), + ).run( + user_prompt="Delegate review.", + run_id="nonrecursive-subagent-run", + ) + + assert result.stop_reason is StopReason.COMPLETED + recursive_result = recursive.requests[1].messages[-1].tool_results[0] + assert json.loads(recursive_result.content)["error"]["code"] == "unknown_tool" + projected = cast( + list[dict[str, object]], + delegated_payload(parent)["children"], + ) + assert projected[0]["untrusted_summary"] == "Recursion was unavailable." + + +@pytest.mark.asyncio +async def test_one_child_timeout_does_not_stop_sibling_or_parent( + tmp_path: Path, +) -> None: + prepare_workspace(tmp_path) + slow = ScriptedProvider( + (stop_response("too late"),), + delay_seconds=0.2, + ) + fast = ScriptedProvider((stop_response("sibling complete"),)) + supervisor, _, _ = supervisor_for( + tmp_path, + profile=profile_for( + child_timeout_seconds=0.03, + batch_timeout_seconds=0.3, + ), + providers=(slow, fast), + ) + parent = parent_provider_for(("Slow review.", "Fast review.")) + + result = await AgentRuntime( + parent, + parent_executor(supervisor), + ).run( + user_prompt="Delegate reviews.", + run_id="timeout-subagent-run", + ) + + assert result.stop_reason is StopReason.COMPLETED + projected = cast( + list[dict[str, object]], + delegated_payload(parent)["children"], + ) + assert [child["status"] for child in projected] == [ + "timed_out", + "completed", + ] + assert projected[1]["untrusted_summary"] == "sibling complete" + + +class CancellationGate: + def __init__(self, expected: int) -> None: + self.expected = expected + self.active = 0 + self.cancelled = 0 + self.reached = asyncio.Event() + self.release = asyncio.Event() + + async def wait(self) -> None: + self.active += 1 + if self.active >= self.expected: + self.reached.set() + try: + await self.release.wait() + except asyncio.CancelledError: + self.cancelled += 1 + raise + finally: + self.active -= 1 + + +class BlockingProvider: + def __init__(self, gate: CancellationGate) -> None: + self._gate = gate + self._capabilities = ProviderCapabilities() + self.requests: list[ModelRequest] = [] + + @property + def capabilities(self) -> ProviderCapabilities: + return self._capabilities + + async def complete(self, request: ModelRequest) -> ModelResponse: + self.requests.append(request) + await self._gate.wait() + return stop_response("released") + + async def stream( + self, + request: ModelRequest, + ) -> AsyncIterator[ProviderStreamEvent]: + yield ResponseCompleted(response=await self.complete(request)) + + +@pytest.mark.asyncio +async def test_parent_cancellation_cancels_both_children( + tmp_path: Path, +) -> None: + prepare_workspace(tmp_path) + gate = CancellationGate(expected=2) + children: tuple[ModelProvider, ...] = ( + BlockingProvider(gate), + BlockingProvider(gate), + ) + supervisor, _, _ = supervisor_for( + tmp_path, + profile=profile_for(), + providers=children, + ) + parent = parent_provider_for(("First review.", "Second review.")) + run = asyncio.create_task( + AgentRuntime( + parent, + parent_executor(supervisor), + ).run( + user_prompt="Delegate reviews.", + run_id="cancelled-subagent-run", + ) + ) + await asyncio.wait_for(gate.reached.wait(), timeout=1) + + run.cancel() + + with pytest.raises(asyncio.CancelledError): + await run + await asyncio.sleep(0) + assert gate.cancelled == 2 + assert gate.active == 0 diff --git a/tests/smoke_test.py b/tests/smoke_test.py index 6bc0248..5d75148 100644 --- a/tests/smoke_test.py +++ b/tests/smoke_test.py @@ -15,6 +15,13 @@ RepairRuntime, ) from mini_code_agent.skills import SkillCatalog +from mini_code_agent.subagents import ( + SubagentAnalysisTool, + SubagentLimits, + SubagentProfile, + SubagentSupervisor, + build_subagent_tools, +) from mini_code_agent.testing import PytestRunner from mini_code_agent.tools import RunTestsTool @@ -24,6 +31,11 @@ def verify_installed_package() -> None: assert RepairActionGuard.__name__ == "RepairActionGuard" assert RepairRuntime.__name__ == "RepairRuntime" assert SkillCatalog.__name__ == "SkillCatalog" + assert SubagentAnalysisTool.__name__ == "SubagentAnalysisTool" + assert SubagentLimits().max_tasks == 4 + assert SubagentProfile.__name__ == "SubagentProfile" + assert SubagentSupervisor.__name__ == "SubagentSupervisor" + assert build_subagent_tools.__name__ == "build_subagent_tools" assert ToolHookRunner.__name__ == "ToolHookRunner" assert McpStdioClient.__name__ == "McpStdioClient" assert McpLimits().max_tools == 32 diff --git a/tests/unit/policy/test_engine.py b/tests/unit/policy/test_engine.py index 4e55cad..ce7606c 100644 --- a/tests/unit/policy/test_engine.py +++ b/tests/unit/policy/test_engine.py @@ -51,6 +51,7 @@ def test_policy_enums_are_stable() -> None: "project", "model", "extension", + "subagent", } @@ -139,6 +140,36 @@ def test_rule_matches_session_and_trust_source() -> None: assert model_request.rule_id == "default-write" +def test_rule_can_match_subagent_without_matching_parent_model() -> None: + engine = PolicyEngine( + rules=( + PolicyRule( + id="allow-subagent-read", + decision=PolicyDecision.ALLOW, + rationale="Bounded child reads are allowed.", + side_effect=SideEffect.READ_ONLY, + trust_source=TrustSource.SUBAGENT, + ), + ) + ) + + child = engine.evaluate( + request( + side_effect=SideEffect.READ_ONLY, + trust_source=TrustSource.SUBAGENT, + ) + ) + parent = engine.evaluate( + request( + side_effect=SideEffect.READ_ONLY, + trust_source=TrustSource.MODEL, + ) + ) + + assert child.rule_id == "allow-subagent-read" + assert parent.rule_id == "default-read-only" + + def test_rule_matches_side_effect_and_tool_glob() -> None: engine = PolicyEngine( rules=( diff --git a/tests/unit/subagents/__init__.py b/tests/unit/subagents/__init__.py new file mode 100644 index 0000000..8b13789 --- /dev/null +++ b/tests/unit/subagents/__init__.py @@ -0,0 +1 @@ + diff --git a/tests/unit/subagents/test_contracts.py b/tests/unit/subagents/test_contracts.py new file mode 100644 index 0000000..a66b7a0 --- /dev/null +++ b/tests/unit/subagents/test_contracts.py @@ -0,0 +1,164 @@ +from __future__ import annotations + +from pathlib import Path +from typing import Literal + +import pytest + +from mini_code_agent.agent.models import AgentLimits +from mini_code_agent.domain.content import ToolCall, ToolResult +from mini_code_agent.policy.models import TrustSource +from mini_code_agent.subagents.contracts import ( + SubagentCompositionError, + SubagentProviderFactory, + SubagentToolFactory, + validate_child_tools, +) +from mini_code_agent.subagents.models import ( + SubagentLimits, + SubagentProfile, +) +from mini_code_agent.tools.base import SideEffect, ToolDefinition + + +def profile_for() -> SubagentProfile: + return SubagentProfile( + profile_id="review", + local_name="delegate_analysis", + description="Run isolated review.", + system_prompt="Review the assigned task.", + tool_names=("read_file", "search_text"), + agent_limits=AgentLimits(max_turns=4, max_tool_calls=8), + limits=SubagentLimits(max_evidence_items=8), + ) + + +def definition( + name: str, + *, + side_effect: SideEffect = SideEffect.READ_ONLY, +) -> ToolDefinition: + return ToolDefinition( + name=name, + description=f"Test {name}.", + input_schema={ + "type": "object", + "properties": {}, + "additionalProperties": False, + }, + side_effect=side_effect, + ) + + +class StubTools: + def __init__( + self, + definitions: tuple[ToolDefinition, ...], + *, + governed: object = True, + trust_source: TrustSource = TrustSource.SUBAGENT, + trust_error: bool = False, + ) -> None: + self._definitions = definitions + self._governed = governed + self._trust_source = trust_source + self._trust_error = trust_error + + @property + def governance_enforced(self) -> object: + return self._governed + + @property + def definitions(self) -> tuple[ToolDefinition, ...]: + return self._definitions + + def trust_source_for(self, tool_name: str) -> TrustSource: + if self._trust_error: + raise RuntimeError("secret implementation failure") + assert tool_name + return self._trust_source + + async def execute(self, call: ToolCall) -> ToolResult: + return ToolResult(tool_call_id=call.id, content="unused") + + +def valid_tools() -> StubTools: + return StubTools((definition("read_file"), definition("search_text"))) + + +def test_validate_child_tools_accepts_exact_governed_subagent_contract() -> None: + validate_child_tools(profile_for(), valid_tools()) + + +@pytest.mark.parametrize( + "tools", + [ + StubTools( + ( + definition("read_file"), + definition("search_text"), + definition("extra"), + ) + ), + StubTools((definition("read_file"),)), + StubTools((definition("search_text"), definition("read_file"))), + StubTools( + ( + definition("read_file"), + definition("search_text", side_effect=SideEffect.WRITE), + ) + ), + StubTools( + (definition("read_file"), definition("search_text")), + governed=False, + ), + StubTools( + (definition("read_file"), definition("search_text")), + governed=1, + ), + StubTools( + (definition("read_file"), definition("search_text")), + trust_source=TrustSource.MODEL, + ), + StubTools( + (definition("read_file"), definition("search_text")), + trust_error=True, + ), + ], +) +def test_validate_child_tools_rejects_authority_drift(tools: StubTools) -> None: + with pytest.raises(SubagentCompositionError) as caught: + validate_child_tools(profile_for(), tools) + + assert str(caught.value) == "Subagent capabilities did not match the host profile." + assert "secret" not in str(caught.value) + + +def test_factory_protocols_define_bounded_inputs() -> None: + class ProviderFactory: + def create(self, profile: SubagentProfile, child_id: str) -> object: + return (profile.profile_id, child_id) + + class ToolFactory: + def create(self, profile: SubagentProfile, workspace_root: Path) -> StubTools: + assert profile.profile_id == "review" + assert workspace_root.is_absolute() + return valid_tools() + + provider_factory: SubagentProviderFactory = ProviderFactory() # type: ignore[assignment] + tool_factory: SubagentToolFactory = ToolFactory() + + assert provider_factory is not None + assert tool_factory.create(profile_for(), Path.cwd().resolve()).definitions + + +def test_governance_property_is_literal_true_in_public_protocol() -> None: + class LiteralTools(StubTools): + @property + def governance_enforced(self) -> Literal[True]: + return True + + validate_child_tools( + profile_for(), + LiteralTools((definition("read_file"), definition("search_text"))), + ) diff --git a/tests/unit/subagents/test_events.py b/tests/unit/subagents/test_events.py new file mode 100644 index 0000000..f87fa7a --- /dev/null +++ b/tests/unit/subagents/test_events.py @@ -0,0 +1,144 @@ +from __future__ import annotations + +from collections.abc import Callable + +import pytest +from pydantic import TypeAdapter, ValidationError + +from mini_code_agent.providers.base import TokenUsage +from mini_code_agent.subagents.events import ( + RecordingSubagentEventSink, + SubagentBatchCompleted, + SubagentBatchStarted, + SubagentCompleted, + SubagentEvent, + SubagentStarted, +) +from mini_code_agent.subagents.models import SubagentStatus + + +def completed_event() -> SubagentCompleted: + return SubagentCompleted( + parent_tool_call_id="parent-1", + profile_id="review", + child_id="child-1", + ordinal=0, + status=SubagentStatus.COMPLETED, + duration_ms=12, + turns=2, + tool_calls=1, + usage=TokenUsage(input_tokens=10, output_tokens=3), + result_sha256="a" * 64, + ) + + +def test_subagent_events_omit_task_prompt_summary_and_results() -> None: + event = completed_event() + + assert set(event.model_dump()) == { + "event_id", + "timestamp", + "type", + "parent_tool_call_id", + "profile_id", + "child_id", + "ordinal", + "status", + "duration_ms", + "turns", + "tool_calls", + "usage", + "result_sha256", + } + payload = event.model_dump_json() + assert "task" not in payload + assert "prompt" not in payload + assert "summary" not in payload + assert "content" not in payload + + +def test_event_union_round_trips_all_lifecycle_types() -> None: + events: tuple[SubagentEvent, ...] = ( + SubagentBatchStarted( + parent_tool_call_id="parent-1", + profile_id="review", + task_count=1, + ), + SubagentStarted( + parent_tool_call_id="parent-1", + profile_id="review", + child_id="child-1", + ordinal=0, + ), + completed_event(), + SubagentBatchCompleted( + parent_tool_call_id="parent-1", + profile_id="review", + duration_ms=14, + completed=1, + stopped=0, + timed_out=0, + failed=0, + result_sha256="b" * 64, + ), + ) + adapter = TypeAdapter[SubagentEvent](SubagentEvent) + + assert tuple(adapter.validate_json(item.model_dump_json()) for item in events) == events + + +def test_recording_sink_preserves_event_order() -> None: + sink = RecordingSubagentEventSink() + started = SubagentStarted( + parent_tool_call_id="parent-1", + profile_id="review", + child_id="child-1", + ordinal=0, + ) + completed = completed_event() + + sink.publish(started) + sink.publish(completed) + + assert sink.events == [started, completed] + + +@pytest.mark.parametrize( + "factory", + [ + lambda: SubagentBatchStarted( + parent_tool_call_id="x" * 129, + profile_id="review", + task_count=1, + ), + lambda: SubagentBatchStarted( + parent_tool_call_id="parent-1", + profile_id="review", + task_count=5, + ), + lambda: SubagentStarted( + parent_tool_call_id="parent-1", + profile_id="review", + child_id="../invalid", + ordinal=0, + ), + lambda: SubagentCompleted.model_validate( + completed_event().model_dump() | {"duration_ms": 3_700_001} + ), + lambda: SubagentBatchCompleted( + parent_tool_call_id="parent-1", + profile_id="review", + duration_ms=1, + completed=5, + stopped=0, + timed_out=0, + failed=0, + result_sha256="b" * 64, + ), + ], +) +def test_events_reject_unbounded_or_invalid_metadata( + factory: Callable[[], object], +) -> None: + with pytest.raises(ValidationError): + factory() diff --git a/tests/unit/subagents/test_evidence.py b/tests/unit/subagents/test_evidence.py new file mode 100644 index 0000000..93b6e02 --- /dev/null +++ b/tests/unit/subagents/test_evidence.py @@ -0,0 +1,174 @@ +from __future__ import annotations + +import json +from hashlib import sha256 + +import pytest + +from mini_code_agent.agent.models import AgentResult, StopReason +from mini_code_agent.domain.content import ToolCall, ToolResult +from mini_code_agent.domain.messages import Message, MessageRole +from mini_code_agent.providers.base import TokenUsage +from mini_code_agent.subagents.evidence import ( + SubagentEvidenceError, + extract_subagent_evidence, + subagent_result_sha256, +) +from mini_code_agent.subagents.models import SubagentEvidenceItem + + +def agent_result_with_tools( + calls: tuple[tuple[str, str, str, bool], ...], +) -> AgentResult: + tool_calls = tuple( + ToolCall(id=call_id, name=tool_name, arguments={}) for call_id, tool_name, _, _ in calls + ) + results = tuple( + ToolResult(tool_call_id=call_id, content=content, is_error=is_error) + for call_id, _, content, is_error in calls + ) + return AgentResult( + run_id="subagent-child-1", + messages=( + Message.user_text("task"), + Message(role=MessageRole.ASSISTANT, content=tool_calls), + Message(role=MessageRole.USER, content=results), + Message.assistant_text("done"), + ), + stop_reason=StopReason.COMPLETED, + turns=2, + tool_calls=len(calls), + usage=TokenUsage(input_tokens=10, output_tokens=2), + final_text="done", + ) + + +def test_extract_evidence_returns_only_bounded_hash_metadata() -> None: + secret = "do-not-copy-result" + result = agent_result_with_tools( + ( + ("call-1", "read_file", secret, False), + ("call-2", "search_text", "two", True), + ) + ) + + evidence = extract_subagent_evidence(result, max_items=2) + + assert [item.tool_name for item in evidence] == ["read_file", "search_text"] + assert evidence[0].content_chars == len(secret) + assert evidence[0].content_sha256 == sha256(secret.encode()).hexdigest() + assert evidence[1].is_error is True + assert secret not in json.dumps([item.model_dump() for item in evidence]) + + +def test_extract_evidence_returns_empty_for_tool_free_result() -> None: + result = AgentResult( + run_id="subagent-child-1", + messages=(Message.user_text("task"), Message.assistant_text("done")), + stop_reason=StopReason.COMPLETED, + turns=1, + tool_calls=0, + usage=TokenUsage(), + final_text="done", + ) + + assert extract_subagent_evidence(result, max_items=0) == () + + +@pytest.mark.parametrize( + "messages", + [ + ( + Message.user_text("task"), + Message( + role=MessageRole.USER, + content=(ToolResult(tool_call_id="missing", content="result"),), + ), + ), + ( + Message.user_text("task"), + Message( + role=MessageRole.ASSISTANT, + content=( + ToolCall(id="duplicate", name="read_file", arguments={}), + ToolCall(id="duplicate", name="search_text", arguments={}), + ), + ), + ), + ( + Message.user_text("task"), + Message( + role=MessageRole.ASSISTANT, + content=(ToolCall(id="call-1", name="read_file", arguments={}),), + ), + ), + ( + Message.user_text("task"), + Message( + role=MessageRole.ASSISTANT, + content=(ToolCall(id="call-1", name="read_file", arguments={}),), + ), + Message( + role=MessageRole.USER, + content=( + ToolResult(tool_call_id="call-1", content="one"), + ToolResult(tool_call_id="call-1", content="two"), + ), + ), + ), + ], +) +def test_extract_evidence_rejects_malformed_correlation( + messages: tuple[Message, ...], +) -> None: + result = AgentResult( + run_id="subagent-child-1", + messages=messages, + stop_reason=StopReason.INVALID_RESPONSE, + turns=1, + tool_calls=1, + usage=TokenUsage(), + error="invalid", + ) + + with pytest.raises(SubagentEvidenceError): + extract_subagent_evidence(result, max_items=4) + + +def test_extract_evidence_rejects_budget_overflow() -> None: + result = agent_result_with_tools( + ( + ("call-1", "read_file", "one", False), + ("call-2", "search_text", "two", False), + ) + ) + + with pytest.raises(SubagentEvidenceError): + extract_subagent_evidence(result, max_items=1) + + +def test_result_hash_is_canonical_and_rejects_non_finite_values() -> None: + left = SubagentEvidenceItem( + tool_call_id="call-1", + tool_name="read_file", + is_error=False, + content_chars=3, + content_sha256="a" * 64, + ) + right = SubagentEvidenceItem.model_validate( + { + "content_sha256": "a" * 64, + "content_chars": 3, + "is_error": False, + "tool_name": "read_file", + "tool_call_id": "call-1", + } + ) + + assert subagent_result_sha256(left) == subagent_result_sha256(right) + + +def test_public_evidence_error_is_static() -> None: + error = SubagentEvidenceError() + + assert str(error) == "Subagent transcript evidence was invalid." diff --git a/tests/unit/subagents/test_models.py b/tests/unit/subagents/test_models.py new file mode 100644 index 0000000..fc7f763 --- /dev/null +++ b/tests/unit/subagents/test_models.py @@ -0,0 +1,305 @@ +from __future__ import annotations + +from collections.abc import Mapping + +import pytest +from pydantic import ValidationError + +from mini_code_agent.agent.models import AgentLimits, StopReason +from mini_code_agent.providers.base import TokenUsage +from mini_code_agent.subagents.models import ( + SubagentBatchResult, + SubagentChildResult, + SubagentErrorCode, + SubagentEvidenceItem, + SubagentLimits, + SubagentProfile, + SubagentStatus, +) + + +def limits_for(**changes: object) -> SubagentLimits: + values: dict[str, object] = { + "max_tasks": 4, + "max_concurrency": 2, + "max_task_chars": 4_000, + "child_timeout_seconds": 120, + "batch_timeout_seconds": 300, + "max_summary_chars": 8_000, + "max_evidence_items": 64, + "max_result_bytes": 131_072, + } + values.update(changes) + return SubagentLimits.model_validate(values) + + +def profile_for( + *, + tool_names: tuple[str, ...] = ("read_file", "search_text"), + agent_limits: AgentLimits | None = None, + limits: SubagentLimits | None = None, + max_tasks: int | None = None, + max_concurrency: int | None = None, +) -> SubagentProfile: + active_limits = limits or limits_for( + **( + { + key: value + for key, value in { + "max_tasks": max_tasks, + "max_concurrency": max_concurrency, + }.items() + if value is not None + } + ) + ) + return SubagentProfile( + profile_id="review", + local_name="delegate_analysis", + description="Run isolated read-only code analysis.", + system_prompt="Inspect only the assigned task and return a concise summary.", + tool_names=tool_names, + agent_limits=agent_limits + or AgentLimits( + max_turns=8, + max_tool_calls=32, + provider_timeout_seconds=30, + tool_timeout_seconds=10, + ), + limits=active_limits, + ) + + +def evidence_for( + *, + call_id: str = "call-1", + tool_name: str = "read_file", +) -> SubagentEvidenceItem: + return SubagentEvidenceItem( + tool_call_id=call_id, + tool_name=tool_name, + is_error=False, + content_chars=12, + content_sha256="a" * 64, + ) + + +def child_for( + *, + child_id: str = "child-1", + ordinal: int = 0, + status: SubagentStatus = SubagentStatus.COMPLETED, + stop_reason: StopReason | None = StopReason.COMPLETED, + error_code: SubagentErrorCode | None = None, + error_message: str | None = None, +) -> SubagentChildResult: + return SubagentChildResult( + child_id=child_id, + ordinal=ordinal, + profile_id="review", + status=status, + stop_reason=stop_reason, + turns=1 if stop_reason is not None else 0, + tool_calls=1 if stop_reason is not None else 0, + usage=TokenUsage(input_tokens=10, output_tokens=2), + untrusted_summary="Review complete." if stop_reason is not None else None, + evidence=(evidence_for(call_id=f"call-{ordinal + 1}"),) if stop_reason is not None else (), + error_code=error_code, + error_message=error_message, + result_sha256="b" * 64, + ) + + +def test_analysis_profile_is_exact_frozen_and_bounded() -> None: + profile = profile_for(max_tasks=3, max_concurrency=2) + + assert profile.mode == "analysis" + assert profile.tool_names == ("read_file", "search_text") + assert profile.limits.max_tasks == 3 + assert profile.limits.max_concurrency == 2 + with pytest.raises(ValidationError): + profile.tool_names = ("write_file",) # type: ignore[misc] + + +@pytest.mark.parametrize( + "tool_names", + [ + (), + ("read_file", "read_file"), + ("delegate_analysis",), + ("delegate_other",), + ("Invalid-Name",), + ], +) +def test_analysis_profile_rejects_empty_duplicate_or_recursive_tools( + tool_names: tuple[str, ...], +) -> None: + with pytest.raises(ValidationError): + profile_for(tool_names=tool_names) + + +@pytest.mark.parametrize( + "changes", + [ + {"max_tasks": 5}, + {"max_concurrency": 5}, + {"max_tasks": 2, "max_concurrency": 3}, + {"child_timeout_seconds": 601}, + {"batch_timeout_seconds": 901}, + {"child_timeout_seconds": 301, "batch_timeout_seconds": 300}, + {"max_task_chars": 20_001}, + {"max_summary_chars": 32_001}, + {"max_evidence_items": 257}, + {"max_result_bytes": 1_048_577}, + ], +) +def test_limits_reject_invalid_or_inconsistent_values( + changes: Mapping[str, object], +) -> None: + with pytest.raises(ValidationError): + limits_for(**changes) + + +def test_profile_rejects_agent_budget_larger_than_subagent_contract() -> None: + with pytest.raises(ValidationError): + profile_for( + agent_limits=AgentLimits(max_turns=33, max_tool_calls=32), + ) + with pytest.raises(ValidationError): + profile_for( + agent_limits=AgentLimits(max_turns=8, max_tool_calls=65), + limits=limits_for(max_evidence_items=64), + ) + + +def test_profile_snapshots_tool_names_and_rejects_nul_prompt() -> None: + names = ["read_file", "search_text"] + profile = SubagentProfile.model_validate(profile_for().model_dump() | {"tool_names": names}) + names.clear() + + assert profile.tool_names == ("read_file", "search_text") + with pytest.raises(ValidationError): + SubagentProfile.model_validate(profile.model_dump() | {"system_prompt": "unsafe\0prompt"}) + + +def test_evidence_is_frozen_bounded_metadata_only() -> None: + evidence = evidence_for() + + assert evidence.content_chars == 12 + assert set(evidence.model_dump()) == { + "tool_call_id", + "tool_name", + "is_error", + "content_chars", + "content_sha256", + } + with pytest.raises(ValidationError): + evidence.content_chars = 13 # type: ignore[misc] + with pytest.raises(ValidationError): + SubagentEvidenceItem.model_validate(evidence.model_dump() | {"content_sha256": "invalid"}) + + +@pytest.mark.parametrize( + ("status", "stop_reason", "error_code", "error_message"), + [ + (SubagentStatus.COMPLETED, None, None, None), + (SubagentStatus.STOPPED, None, None, None), + ( + SubagentStatus.TIMED_OUT, + StopReason.PROVIDER_TIMEOUT, + SubagentErrorCode.CHILD_TIMEOUT, + "Subagent timed out.", + ), + (SubagentStatus.FAILED, None, None, None), + ( + SubagentStatus.BATCH_TIMED_OUT, + None, + SubagentErrorCode.CHILD_FAILED, + "wrong code", + ), + ], +) +def test_child_result_rejects_inconsistent_status_fields( + status: SubagentStatus, + stop_reason: StopReason | None, + error_code: SubagentErrorCode | None, + error_message: str | None, +) -> None: + with pytest.raises(ValidationError): + child_for( + status=status, + stop_reason=stop_reason, + error_code=error_code, + error_message=error_message, + ) + + +def test_batch_result_factory_counts_statuses_and_hashes_projection() -> None: + children = ( + child_for(ordinal=0), + child_for( + child_id="child-2", + ordinal=1, + status=SubagentStatus.TIMED_OUT, + stop_reason=None, + error_code=SubagentErrorCode.CHILD_TIMEOUT, + error_message="Subagent timed out.", + ), + child_for( + child_id="child-3", + ordinal=2, + status=SubagentStatus.FAILED, + stop_reason=None, + error_code=SubagentErrorCode.CHILD_FAILED, + error_message="Subagent failed.", + ), + ) + + batch = SubagentBatchResult.from_children( + profile_id="review", + children=children, + duration_ms=10, + ) + + assert batch.completed == 1 + assert batch.stopped == 0 + assert batch.timed_out == 1 + assert batch.failed == 1 + assert len(batch.result_sha256) == 64 + assert batch.children == children + + +def test_batch_result_rejects_wrong_counts_order_or_profile() -> None: + valid = SubagentBatchResult.from_children( + profile_id="review", + children=(child_for(),), + duration_ms=10, + ) + with pytest.raises(ValidationError): + SubagentBatchResult.model_validate(valid.model_dump() | {"completed": 0}) + with pytest.raises(ValidationError): + SubagentBatchResult.from_children( + profile_id="review", + children=(child_for(ordinal=1),), + duration_ms=10, + ) + with pytest.raises(ValidationError): + SubagentBatchResult.from_children( + profile_id="other", + children=(child_for(),), + duration_ms=10, + ) + + +def test_result_models_reject_unbounded_summary_evidence_and_ids() -> None: + with pytest.raises(ValidationError): + child_for(child_id="x" * 97) + with pytest.raises(ValidationError): + SubagentChildResult.model_validate( + child_for().model_dump() | {"untrusted_summary": "x" * 32_001} + ) + with pytest.raises(ValidationError): + SubagentChildResult.model_validate( + child_for().model_dump() + | {"evidence": tuple(evidence_for(call_id=f"call-{i}") for i in range(257))} + ) diff --git a/tests/unit/subagents/test_supervisor.py b/tests/unit/subagents/test_supervisor.py new file mode 100644 index 0000000..1c1d703 --- /dev/null +++ b/tests/unit/subagents/test_supervisor.py @@ -0,0 +1,656 @@ +from __future__ import annotations + +import asyncio +from collections import deque +from collections.abc import AsyncIterator, Callable +from pathlib import Path + +import pytest + +from mini_code_agent.agent.models import AgentLimits, StopReason +from mini_code_agent.domain.content import ToolCall, ToolResult +from mini_code_agent.domain.messages import Message +from mini_code_agent.policy.approval import StaticApprovalHandler +from mini_code_agent.policy.engine import PolicyEngine +from mini_code_agent.policy.executor import GovernedToolExecutor +from mini_code_agent.policy.models import SessionMode, TrustSource +from mini_code_agent.providers.base import ( + FinishReason, + ModelProvider, + ModelRequest, + ModelResponse, + ProviderCapabilities, + ProviderStreamEvent, + ResponseCompleted, +) +from mini_code_agent.providers.fake import ScriptedProvider +from mini_code_agent.subagents.contracts import SubagentCompositionError +from mini_code_agent.subagents.events import ( + RecordingSubagentEventSink, + SubagentBatchCompleted, + SubagentBatchStarted, + SubagentCompleted, + SubagentStarted, +) +from mini_code_agent.subagents.models import ( + SubagentBatchResult, + SubagentError, + SubagentErrorCode, + SubagentLimits, + SubagentProfile, + SubagentStatus, +) +from mini_code_agent.subagents.supervisor import SubagentSupervisor +from mini_code_agent.tools.base import SideEffect, ToolDefinition, ToolExecutor +from mini_code_agent.tools.registry import ToolRegistry + + +class ReadOnlyTool: + def __init__(self, name: str) -> None: + self._definition = ToolDefinition( + name=name, + description=f"Test {name}.", + input_schema={ + "type": "object", + "properties": {}, + "additionalProperties": False, + }, + side_effect=SideEffect.READ_ONLY, + ) + + @property + def definition(self) -> ToolDefinition: + return self._definition + + async def execute(self, call: ToolCall) -> ToolResult: + return ToolResult(tool_call_id=call.id, content=f"{call.name} result") + + +def governed_tools( + *, + trust_source: TrustSource = TrustSource.SUBAGENT, +) -> GovernedToolExecutor: + return GovernedToolExecutor( + ToolRegistry((ReadOnlyTool("read_file"), ReadOnlyTool("search_text"))), + policy=PolicyEngine(), + approval=StaticApprovalHandler(approved=False), + session_mode=SessionMode.NON_INTERACTIVE, + trust_source=trust_source, + ) + + +def final_response( + text: str = "review complete", + *, + finish_reason: FinishReason = FinishReason.STOP, +) -> ModelResponse: + return ModelResponse( + message=Message.assistant_text(text), + finish_reason=finish_reason, + ) + + +class ProviderFactory: + def __init__( + self, + providers: tuple[ModelProvider, ...], + *, + error: Exception | None = None, + ) -> None: + self.providers = deque(providers) + self.error = error + self.calls: list[tuple[str, str]] = [] + + def create( + self, + profile: SubagentProfile, + child_id: str, + ) -> ModelProvider: + self.calls.append((profile.profile_id, child_id)) + if self.error is not None: + raise self.error + return self.providers.popleft() + + +class ToolFactory: + def __init__( + self, + tools: tuple[ToolExecutor, ...], + *, + error: Exception | None = None, + ) -> None: + self.tools = deque(tools) + self.error = error + self.calls: list[tuple[str, Path]] = [] + + def create( + self, + profile: SubagentProfile, + workspace_root: Path, + ) -> ToolExecutor: + self.calls.append((profile.profile_id, workspace_root)) + if self.error is not None: + raise self.error + return self.tools.popleft() + + +class ConcurrencyGate: + def __init__(self, expected: int, *, auto_release: bool = True) -> None: + self.expected = expected + self.auto_release = auto_release + self.entered = 0 + self.active = 0 + self.peak = 0 + self.cancelled = 0 + self.reached = asyncio.Event() + self.release = asyncio.Event() + + async def wait(self) -> None: + self.entered += 1 + self.active += 1 + self.peak = max(self.peak, self.active) + if self.entered >= self.expected: + self.reached.set() + if self.auto_release: + self.release.set() + try: + await self.release.wait() + except asyncio.CancelledError: + self.cancelled += 1 + raise + finally: + self.active -= 1 + + +class GatedProvider: + def __init__( + self, + gate: ConcurrencyGate, + *, + result: str, + delay_after_gate: float = 0, + ) -> None: + self._gate = gate + self._result = result + self._delay_after_gate = delay_after_gate + self.requests: list[ModelRequest] = [] + self._capabilities = ProviderCapabilities() + + @property + def capabilities(self) -> ProviderCapabilities: + return self._capabilities + + async def complete(self, request: ModelRequest) -> ModelResponse: + self.requests.append(request) + await self._gate.wait() + if self._delay_after_gate: + await asyncio.sleep(self._delay_after_gate) + return final_response(self._result) + + async def stream( + self, + request: ModelRequest, + ) -> AsyncIterator[ProviderStreamEvent]: + yield ResponseCompleted(response=await self.complete(request)) + + +def profile_for(**limit_changes: object) -> SubagentProfile: + limits: dict[str, object] = { + "max_tasks": 4, + "max_concurrency": 2, + "max_evidence_items": 8, + "child_timeout_seconds": 1, + "batch_timeout_seconds": 3, + } + limits.update(limit_changes) + return SubagentProfile( + profile_id="review", + local_name="delegate_analysis", + description="Run isolated review.", + system_prompt="Review only the assigned task.", + tool_names=("read_file", "search_text"), + agent_limits=AgentLimits( + max_turns=4, + max_tool_calls=8, + provider_timeout_seconds=1, + tool_timeout_seconds=1, + ), + limits=SubagentLimits.model_validate(limits), + ) + + +def supervisor_for( + tmp_path: Path, + *, + providers: tuple[ModelProvider, ...], + tools: tuple[ToolExecutor, ...] | None = None, + events: RecordingSubagentEventSink | None = None, + child_ids: tuple[str, ...] = ("child-1",), + profile: SubagentProfile | None = None, + monotonic: Callable[[], float] | None = None, +) -> tuple[SubagentSupervisor, ProviderFactory, ToolFactory]: + provider_factory = ProviderFactory(providers) + tool_factory = ToolFactory(tools or tuple(governed_tools() for _ in providers)) + ids = iter(child_ids) + supervisor = SubagentSupervisor( + profile or profile_for(), + workspace_root=tmp_path, + provider_factory=provider_factory, + tool_factory=tool_factory, + events=events, + id_factory=lambda: next(ids), + monotonic=monotonic or (lambda: 1.0), + ) + return supervisor, provider_factory, tool_factory + + +@pytest.mark.asyncio +async def test_one_child_gets_fresh_context_exact_tools_and_bounded_result( + tmp_path: Path, +) -> None: + provider = ScriptedProvider((final_response(),)) + events = RecordingSubagentEventSink() + supervisor, provider_factory, tool_factory = supervisor_for( + tmp_path, + providers=(provider,), + events=events, + ) + + batch = await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("Inspect parser bounds.",), + ) + + assert provider.requests[0].system_prompt == supervisor.profile.system_prompt + assert provider.requests[0].messages == (Message.user_text("Inspect parser bounds."),) + assert tuple(item.name for item in provider.requests[0].tools) == ( + "read_file", + "search_text", + ) + assert batch.children[0].untrusted_summary == "review complete" + assert batch.children[0].status is SubagentStatus.COMPLETED + assert batch.children[0].stop_reason is StopReason.COMPLETED + assert batch.completed == 1 + assert provider_factory.calls == [("review", "child-1")] + assert tool_factory.calls == [("review", tmp_path.resolve())] + assert [type(item) for item in events.events] == [ + SubagentBatchStarted, + SubagentStarted, + SubagentCompleted, + SubagentBatchCompleted, + ] + + +@pytest.mark.asyncio +async def test_non_completed_agent_result_is_stopped_not_failed( + tmp_path: Path, +) -> None: + provider = ScriptedProvider( + (final_response("limit reached", finish_reason=FinishReason.MAX_TOKENS),) + ) + supervisor, _, _ = supervisor_for(tmp_path, providers=(provider,)) + + batch = await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("Inspect parser.",), + ) + + child = batch.children[0] + assert child.status is SubagentStatus.STOPPED + assert child.stop_reason is StopReason.PROVIDER_LIMIT + assert batch.stopped == 1 + + +@pytest.mark.asyncio +@pytest.mark.parametrize("factory_name", ["provider", "tools"]) +async def test_factory_failure_is_static_and_starts_no_provider( + tmp_path: Path, + factory_name: str, +) -> None: + provider = ScriptedProvider((final_response(),)) + provider_factory = ProviderFactory( + (provider,), + error=RuntimeError("secret provider") if factory_name == "provider" else None, + ) + tool_factory = ToolFactory( + (governed_tools(),), + error=RuntimeError("secret tools") if factory_name == "tools" else None, + ) + events = RecordingSubagentEventSink() + supervisor = SubagentSupervisor( + profile_for(), + workspace_root=tmp_path, + provider_factory=provider_factory, + tool_factory=tool_factory, + events=events, + id_factory=lambda: "child-1", + ) + + with pytest.raises(SubagentCompositionError) as caught: + await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("Inspect parser.",), + ) + + assert str(caught.value) == "Subagent capabilities did not match the host profile." + assert provider.requests == [] + assert events.events == [] + + +@pytest.mark.asyncio +async def test_all_children_are_composed_before_any_provider_call( + tmp_path: Path, +) -> None: + shared = ScriptedProvider((final_response(), final_response())) + supervisor, _, _ = supervisor_for( + tmp_path, + providers=(shared, shared), + child_ids=("child-1", "child-2"), + ) + + with pytest.raises(SubagentCompositionError): + await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("one", "two"), + ) + + assert shared.requests == [] + + +@pytest.mark.asyncio +async def test_duplicate_child_ids_fail_composition_before_provider_call( + tmp_path: Path, +) -> None: + providers = ( + ScriptedProvider((final_response("first"),)), + ScriptedProvider((final_response("second"),)), + ) + supervisor, _, _ = supervisor_for( + tmp_path, + providers=providers, + child_ids=("same-child", "same-child"), + ) + + with pytest.raises(SubagentCompositionError): + await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("one", "two"), + ) + + assert all(provider.requests == [] for provider in providers) + + +@pytest.mark.asyncio +async def test_duplicate_tasks_are_rejected_before_child_composition( + tmp_path: Path, +) -> None: + provider = ScriptedProvider((final_response(),)) + supervisor, provider_factory, tool_factory = supervisor_for( + tmp_path, + providers=(provider,), + ) + + with pytest.raises(SubagentError) as caught: + await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("same task", "same task"), + ) + + assert caught.value.code is SubagentErrorCode.INVALID_BATCH + assert provider_factory.calls == [] + assert tool_factory.calls == [] + assert provider.requests == [] + + +@pytest.mark.asyncio +async def test_event_sink_failure_does_not_replace_child_result( + tmp_path: Path, +) -> None: + class FailingSink: + def publish(self, event: object) -> None: + del event + raise RuntimeError("sink failed") + + provider = ScriptedProvider((final_response(),)) + provider_factory = ProviderFactory((provider,)) + tool_factory = ToolFactory((governed_tools(),)) + supervisor = SubagentSupervisor( + profile_for(), + workspace_root=tmp_path, + provider_factory=provider_factory, + tool_factory=tool_factory, + events=FailingSink(), # type: ignore[arg-type] + id_factory=lambda: "child-1", + ) + + batch = await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("Inspect parser.",), + ) + + assert batch.completed == 1 + + +def test_supervisor_rejects_invalid_workspace_root(tmp_path: Path) -> None: + missing = tmp_path / "missing" + + with pytest.raises(ValueError): + SubagentSupervisor( + profile_for(), + workspace_root=missing, + provider_factory=ProviderFactory(()), + tool_factory=ToolFactory(()), + ) + + +@pytest.mark.asyncio +async def test_batch_runs_children_concurrently_and_preserves_input_order( + tmp_path: Path, +) -> None: + gate = ConcurrencyGate(expected=2) + providers: tuple[ModelProvider, ...] = ( + GatedProvider(gate, result="first", delay_after_gate=0.05), + GatedProvider(gate, result="second"), + ) + supervisor, _, _ = supervisor_for( + tmp_path, + providers=providers, + child_ids=("child-1", "child-2"), + ) + + result = await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("slow first", "fast second"), + ) + + assert gate.peak == 2 + assert [child.untrusted_summary for child in result.children] == [ + "first", + "second", + ] + assert gate.active == 0 + + +@pytest.mark.asyncio +async def test_child_timeout_does_not_cancel_completed_sibling( + tmp_path: Path, +) -> None: + providers: tuple[ModelProvider, ...] = ( + ScriptedProvider((final_response("late"),), delay_seconds=0.2), + ScriptedProvider((final_response("done"),)), + ) + supervisor, _, _ = supervisor_for( + tmp_path, + providers=providers, + child_ids=("child-1", "child-2"), + profile=profile_for( + child_timeout_seconds=0.05, + batch_timeout_seconds=0.3, + ), + ) + + result = await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("slow", "fast"), + ) + + assert [child.status for child in result.children] == [ + SubagentStatus.TIMED_OUT, + SubagentStatus.COMPLETED, + ] + assert result.timed_out == 1 + assert result.completed == 1 + + +@pytest.mark.asyncio +async def test_ordinary_child_projection_failure_does_not_cancel_sibling( + tmp_path: Path, +) -> None: + providers: tuple[ModelProvider, ...] = ( + ScriptedProvider((final_response("invalid\0summary"),)), + ScriptedProvider((final_response("done"),)), + ) + supervisor, _, _ = supervisor_for( + tmp_path, + providers=providers, + child_ids=("child-1", "child-2"), + ) + + result = await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("invalid", "valid"), + ) + + assert [child.status for child in result.children] == [ + SubagentStatus.FAILED, + SubagentStatus.COMPLETED, + ] + assert result.failed == 1 + assert result.completed == 1 + + +@pytest.mark.asyncio +async def test_max_concurrency_one_never_overlaps_children( + tmp_path: Path, +) -> None: + gate = ConcurrencyGate(expected=1) + providers: tuple[ModelProvider, ...] = ( + GatedProvider(gate, result="first"), + GatedProvider(gate, result="second"), + ) + supervisor, _, _ = supervisor_for( + tmp_path, + providers=providers, + child_ids=("child-1", "child-2"), + profile=profile_for(max_concurrency=1), + ) + + result = await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("first", "second"), + ) + + assert result.completed == 2 + assert gate.peak == 1 + assert gate.active == 0 + + +@pytest.mark.asyncio +async def test_batch_timeout_marks_every_unfinished_ordinal( + tmp_path: Path, +) -> None: + gate = ConcurrencyGate(expected=2, auto_release=False) + providers: tuple[ModelProvider, ...] = ( + GatedProvider(gate, result="first"), + GatedProvider(gate, result="second"), + ScriptedProvider((final_response("never started"),)), + ) + supervisor, _, _ = supervisor_for( + tmp_path, + providers=providers, + child_ids=("child-1", "child-2", "child-3"), + profile=profile_for( + max_concurrency=2, + child_timeout_seconds=0.08, + batch_timeout_seconds=0.08, + ), + ) + + result = await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("first", "second", "waiting"), + ) + + assert [child.ordinal for child in result.children] == [0, 1, 2] + assert all(child.status is SubagentStatus.BATCH_TIMED_OUT for child in result.children) + assert result.timed_out == 3 + assert gate.cancelled == 2 + assert gate.active == 0 + + +@pytest.mark.asyncio +async def test_external_cancellation_cancels_children_and_is_re_raised( + tmp_path: Path, +) -> None: + gate = ConcurrencyGate(expected=2, auto_release=False) + providers: tuple[ModelProvider, ...] = ( + GatedProvider(gate, result="first"), + GatedProvider(gate, result="second"), + ) + supervisor, _, _ = supervisor_for( + tmp_path, + providers=providers, + child_ids=("child-1", "child-2"), + ) + run = asyncio.create_task( + supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("first", "second"), + ) + ) + await asyncio.wait_for(gate.reached.wait(), timeout=1) + + run.cancel() + + with pytest.raises(asyncio.CancelledError): + await run + await asyncio.sleep(0) + assert gate.cancelled == 2 + assert gate.active == 0 + + +@pytest.mark.asyncio +async def test_mixed_batch_counts_and_hash_are_deterministic( + tmp_path: Path, +) -> None: + async def run_once() -> SubagentBatchResult: + providers: tuple[ModelProvider, ...] = ( + ScriptedProvider((final_response("late"),), delay_seconds=0.05), + ScriptedProvider((final_response("invalid\0summary"),)), + ScriptedProvider((final_response("done"),)), + ) + supervisor, _, _ = supervisor_for( + tmp_path, + providers=providers, + child_ids=("child-1", "child-2", "child-3"), + profile=profile_for( + max_concurrency=3, + child_timeout_seconds=0.01, + batch_timeout_seconds=0.2, + ), + ) + return await supervisor.run_batch( + parent_tool_call_id="parent-1", + tasks=("timeout", "fail", "complete"), + ) + + first = await run_once() + second = await run_once() + + assert (first.completed, first.failed, first.timed_out, first.stopped) == ( + 1, + 1, + 1, + 0, + ) + assert first == second diff --git a/tests/unit/subagents/test_tools.py b/tests/unit/subagents/test_tools.py new file mode 100644 index 0000000..ddf135c --- /dev/null +++ b/tests/unit/subagents/test_tools.py @@ -0,0 +1,380 @@ +from __future__ import annotations + +import asyncio +import json +from collections.abc import Mapping +from typing import cast + +import pytest +from pydantic import JsonValue + +import mini_code_agent.subagents as subagent_api +from mini_code_agent.agent.models import AgentLimits, StopReason +from mini_code_agent.domain.content import ToolCall +from mini_code_agent.policy.approval import StaticApprovalHandler +from mini_code_agent.policy.engine import PolicyEngine +from mini_code_agent.policy.executor import GovernedToolExecutor +from mini_code_agent.policy.models import RiskLevel, SessionMode, TrustSource +from mini_code_agent.subagents.contracts import SubagentCompositionError +from mini_code_agent.subagents.models import ( + SubagentBatchResult, + SubagentChildResult, + SubagentError, + SubagentErrorCode, + SubagentLimits, + SubagentProfile, + SubagentStatus, +) +from mini_code_agent.subagents.tools import ( + SubagentAnalysisTool, + build_subagent_tools, +) +from mini_code_agent.tools.base import SideEffect +from mini_code_agent.tools.registry import ToolRegistry + + +def profile_for( + *, + profile_id: str = "review", + local_name: str = "delegate_analysis", + tool_names: tuple[str, ...] = ("read_file", "search_text"), + max_tasks: int = 4, + max_task_chars: int = 100, + max_result_bytes: int = 131_072, +) -> SubagentProfile: + return SubagentProfile( + profile_id=profile_id, + local_name=local_name, + description="Run an isolated code review.", + system_prompt="Review only the assigned task.", + tool_names=tool_names, + agent_limits=AgentLimits(max_turns=4, max_tool_calls=8), + limits=SubagentLimits( + max_tasks=max_tasks, + max_concurrency=min(2, max_tasks), + max_task_chars=max_task_chars, + max_evidence_items=8, + max_result_bytes=max_result_bytes, + ), + ) + + +def child_for( + ordinal: int, + *, + summary: str = "done", +) -> SubagentChildResult: + return SubagentChildResult( + child_id=f"child-{ordinal + 1}", + ordinal=ordinal, + profile_id="review", + status=SubagentStatus.COMPLETED, + stop_reason=StopReason.COMPLETED, + turns=1, + tool_calls=0, + untrusted_summary=summary, + result_sha256="a" * 64, + ) + + +def batch_for( + count: int = 2, + *, + summary: str = "done", +) -> SubagentBatchResult: + return SubagentBatchResult.from_children( + profile_id="review", + children=tuple(child_for(index, summary=summary) for index in range(count)), + duration_ms=10, + ) + + +class FakeSupervisor: + def __init__( + self, + profile: SubagentProfile, + *, + result: object | None = None, + error: BaseException | None = None, + ) -> None: + self.profile = profile + self.result = result if result is not None else batch_for() + self.error = error + self.calls: list[tuple[str, tuple[str, ...]]] = [] + + async def run_batch( + self, + *, + parent_tool_call_id: str, + tasks: tuple[str, ...], + ) -> SubagentBatchResult: + self.calls.append((parent_tool_call_id, tasks)) + if self.error is not None: + raise self.error + return cast(SubagentBatchResult, self.result) + + +def call_for( + *, + tasks: tuple[str, ...] = ("one", "two"), + reason: str = "Independent review.", + name: str = "delegate_analysis", + extra: Mapping[str, JsonValue] | None = None, +) -> ToolCall: + arguments: dict[str, JsonValue] = { + "tasks": list(tasks), + "reason": reason, + } + arguments.update(extra or {}) + return ToolCall( + id="parent-1", + name=name, + arguments=arguments, + ) + + +def error_code(content: str) -> str: + return cast(str, json.loads(content)["error"]["code"]) + + +def test_definition_snapshots_profile_specific_schema() -> None: + profile = profile_for(max_tasks=3, max_task_chars=77) + tool = SubagentAnalysisTool(FakeSupervisor(profile)) + schema = tool.definition.model_dump(mode="json")["input_schema"] + + assert tool.definition.name == "delegate_analysis" + assert tool.definition.description == profile.description + assert tool.definition.side_effect is SideEffect.READ_ONLY + assert schema["properties"]["tasks"]["maxItems"] == 3 + assert schema["properties"]["tasks"]["items"]["maxLength"] == 77 + assert schema["additionalProperties"] is False + + +@pytest.mark.asyncio +async def test_preview_is_read_only_medium_risk_and_bounded() -> None: + tool = SubagentAnalysisTool(FakeSupervisor(profile_for())) + + preview = await tool.preview( + call_for( + tasks=("Inspect parser.", "Inspect serializer."), + reason="Independent review.", + ) + ) + + assert preview.side_effect is SideEffect.READ_ONLY + assert preview.risk is RiskLevel.MEDIUM + assert preview.resources == (".",) + assert "2" in preview.summary + assert preview.reason == "Independent review." + assert "Inspect parser." not in preview.model_dump_json() + + +@pytest.mark.asyncio +async def test_execute_returns_deterministic_bounded_batch_json() -> None: + supervisor = FakeSupervisor(profile_for()) + tool = SubagentAnalysisTool(supervisor) + + first = await tool.execute(call_for()) + second = await tool.execute(call_for()) + payload = json.loads(first.content) + + assert first.is_error is False + assert first.content == second.content + assert payload["content_type"] == "subagent_batch_result" + assert [child["ordinal"] for child in payload["children"]] == [0, 1] + assert supervisor.calls == [ + ("parent-1", ("one", "two")), + ("parent-1", ("one", "two")), + ] + first.content.encode("ascii") + + +@pytest.mark.asyncio +async def test_tool_runs_through_registry_and_governed_policy_path() -> None: + supervisor = FakeSupervisor(profile_for()) + tool = SubagentAnalysisTool(supervisor) + executor = GovernedToolExecutor( + ToolRegistry((tool,)), + policy=PolicyEngine(), + approval=StaticApprovalHandler(approved=False), + session_mode=SessionMode.NON_INTERACTIVE, + trust_source=TrustSource.MODEL, + ) + + result = await executor.execute(call_for()) + duplicate = await executor.execute(call_for(tasks=("same", "same"))) + + assert result.is_error is False + assert error_code(duplicate.content) == "invalid_arguments" + assert supervisor.calls == [("parent-1", ("one", "two"))] + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "call", + [ + call_for(extra={"unexpected": True}), + call_for(tasks=()), + call_for(tasks=("one", "two", "three", "four", "five")), + call_for(tasks=("same", "same")), + call_for(tasks=("x" * 101,)), + call_for(reason=""), + call_for(reason="x" * 501), + call_for(reason="bad\0reason"), + ], +) +async def test_execute_rejects_invalid_arguments(call: ToolCall) -> None: + supervisor = FakeSupervisor(profile_for()) + result = await SubagentAnalysisTool(supervisor).execute(call) + + assert result.is_error is True + assert error_code(result.content) == "invalid_arguments" + assert supervisor.calls == [] + + +@pytest.mark.asyncio +async def test_execute_rejects_wrong_tool_name() -> None: + result = await SubagentAnalysisTool(FakeSupervisor(profile_for())).execute( + call_for(name="other_tool") + ) + + assert result.is_error is True + assert error_code(result.content) == "unknown_tool" + + +@pytest.mark.asyncio +async def test_execute_rejects_oversized_serialized_result() -> None: + profile = profile_for(max_result_bytes=100) + supervisor = FakeSupervisor( + profile, + result=batch_for(summary="x" * 500), + ) + + result = await SubagentAnalysisTool(supervisor).execute(call_for()) + + assert result.is_error is True + assert error_code(result.content) == SubagentErrorCode.RESULT_TOO_LARGE.value + assert "x" * 100 not in result.content + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("failure", "expected_code"), + [ + ( + SubagentError( + SubagentErrorCode.INVALID_BATCH, + "Subagent batch request was invalid.", + ), + SubagentErrorCode.INVALID_BATCH.value, + ), + ( + SubagentCompositionError(), + SubagentErrorCode.COMPOSITION_FAILED.value, + ), + ( + RuntimeError("secret failure"), + SubagentErrorCode.CHILD_FAILED.value, + ), + ], +) +async def test_execute_maps_supervisor_failures_to_static_errors( + failure: Exception, + expected_code: str, +) -> None: + supervisor = FakeSupervisor(profile_for(), error=failure) + + result = await SubagentAnalysisTool(supervisor).execute(call_for()) + + assert result.is_error is True + assert error_code(result.content) == expected_code + assert "secret" not in result.content + + +@pytest.mark.asyncio +async def test_execute_rejects_malformed_supervisor_result() -> None: + supervisor = FakeSupervisor(profile_for(), result={"not": "a batch"}) + + result = await SubagentAnalysisTool(supervisor).execute(call_for()) + + assert result.is_error is True + assert error_code(result.content) == SubagentErrorCode.CHILD_FAILED.value + + +@pytest.mark.asyncio +async def test_execute_re_raises_cancellation() -> None: + supervisor = FakeSupervisor(profile_for(), error=asyncio.CancelledError()) + + with pytest.raises(asyncio.CancelledError): + await SubagentAnalysisTool(supervisor).execute(call_for()) + + +@pytest.mark.parametrize( + "profiles", + [ + ( + profile_for(profile_id="same", local_name="analysis_a"), + profile_for(profile_id="same", local_name="analysis_b"), + ), + ( + profile_for(profile_id="one", local_name="same_name"), + profile_for(profile_id="two", local_name="same_name"), + ), + ( + profile_for(profile_id="one", local_name="analysis_a"), + profile_for( + profile_id="two", + local_name="analysis_b", + tool_names=("analysis_a",), + ), + ), + ], +) +def test_builder_rejects_profile_and_parent_child_name_conflicts( + profiles: tuple[SubagentProfile, SubagentProfile], +) -> None: + supervisors = tuple(FakeSupervisor(profile) for profile in profiles) + + with pytest.raises(ValueError, match="Subagent Tool profiles conflict"): + build_subagent_tools(supervisors) + + +def test_builder_returns_one_distinct_tool_per_profile() -> None: + profiles = ( + profile_for(profile_id="one", local_name="analysis_a"), + profile_for(profile_id="two", local_name="analysis_b"), + ) + + tools = build_subagent_tools(tuple(FakeSupervisor(profile) for profile in profiles)) + + assert [tool.definition.name for tool in tools] == ["analysis_a", "analysis_b"] + assert tools[0].definition is not tools[1].definition + + +def test_package_exports_stable_subagent_api() -> None: + expected = { + "NullSubagentEventSink", + "RecordingSubagentEventSink", + "SubagentAnalysisTool", + "SubagentBatchCompleted", + "SubagentBatchResult", + "SubagentBatchStarted", + "SubagentChildResult", + "SubagentCompleted", + "SubagentCompositionError", + "SubagentError", + "SubagentErrorCode", + "SubagentEvent", + "SubagentEventSink", + "SubagentEvidenceItem", + "SubagentLimits", + "SubagentProfile", + "SubagentProviderFactory", + "SubagentStarted", + "SubagentStatus", + "SubagentSupervisor", + "SubagentToolFactory", + "build_subagent_tools", + } + + assert expected.issubset(set(subagent_api.__all__)) diff --git a/tests/unit/test_package.py b/tests/unit/test_package.py index 4a725b7..1eff080 100644 --- a/tests/unit/test_package.py +++ b/tests/unit/test_package.py @@ -4,7 +4,7 @@ def test_package_exports_release_version() -> None: - assert __version__ == "0.14.0a0" + assert __version__ == "0.15.0a0" def test_package_includes_pep561_marker() -> None: diff --git a/tests/unit/tools/test_runtime_info.py b/tests/unit/tools/test_runtime_info.py index d7f0a93..86956e2 100644 --- a/tests/unit/tools/test_runtime_info.py +++ b/tests/unit/tools/test_runtime_info.py @@ -35,7 +35,7 @@ async def test_runtime_info_returns_safe_structured_data() -> None: payload = json.loads(result.content) assert result.tool_call_id == "call-1" assert result.is_error is False - assert payload["package_version"] == "0.14.0a0" + assert payload["package_version"] == "0.15.0a0" assert payload["python_version"] assert payload["platform"] diff --git a/uv.lock b/uv.lock index e63800d..a44bd0b 100644 --- a/uv.lock +++ b/uv.lock @@ -360,7 +360,7 @@ wheels = [ [[package]] name = "mini-code-agent" -version = "0.14.0a0" +version = "0.15.0a0" source = { editable = "." } dependencies = [ { name = "defusedxml" },