From c7f5930cd892ee605add60d4c974bcec71f9fc4e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 6 Sep 2026 02:59:59 +0530 Subject: [PATCH 001/197] docs: assess engineering team readiness and delivery path --- .../2026-09-06-engineering-team-assessment.md | 183 ++++++++++++++++++ 1 file changed, 183 insertions(+) create mode 100644 docs/audits/2026-09-06-engineering-team-assessment.md diff --git a/docs/audits/2026-09-06-engineering-team-assessment.md b/docs/audits/2026-09-06-engineering-team-assessment.md new file mode 100644 index 0000000..b769781 --- /dev/null +++ b/docs/audits/2026-09-06-engineering-team-assessment.md @@ -0,0 +1,183 @@ +# DevSquad: engineering team assessment and path to usability + +Date: 2026-09-06 + +Baseline: main at fa68f68621dcad166ab24cf4b3db379ecbe819ac, plugin 0.10.0 + +Status: Assessment and proposed delivery plan. This document does not supersede ADR-001 or claim the proposed features are implemented. + +## Verdict + +DevSquad is worth continuing as a personal AI engineering team coordinator. It has useful provider integration code and regression tests, but it is not yet a dependable team that owns engineering outcomes or makes effective use of subscription capacity. + +The intended product has two connected objectives: better engineering through complementary capabilities and independent judgment; and more accepted work from the subscriptions already available. Information access is one specialization alongside architecture, implementation, debugging, design, testing, and review. + +The current product grew around preserving Claude's context. Its hooks, agent prompts, measurement, and capacity messages still express that earlier objective. Those choices explain much of the present mismatch. The next iteration should make a complete, verified engineering task the unit of work and measurement. + +Keep the adapter experience, error taxonomy, compatibility tests, and the accepted shared-core direction. Deliver a small working team before extending platform scope. Five delivery gates below define that path; each requires observable behavior rather than completion checkboxes. + +## Scope and evidence + +Reviewed the local source, complete Git history, relevant local planning and verification records, GitHub branches/releases/workflows, installed plugin metadata and files, scoped DevSquad configuration/telemetry, and current official provider documentation. Earlier in this review, all 177 assertions in nine offline test files passed under Bash 3.2. Those tests establish selected implementation contracts, not live provider compatibility or output superiority. + +GitHub main matches the baseline. The holdout-protocol branch is fully merged. Installed 0.10.0 plugin source matches the current plugin source. GitHub currently has no Actions workflows or published releases; the legacy v1.0 tag references a manifest numbered 0.1.0. Runtime code has not changed since July 7; later commits concern presentation, documentation, and motion assets. + +The local probes inspected CLI help, versions, model listings, and model resolution. They did not execute paid model tasks or benchmark engineering quality. Grok reported expired authentication. The cached model list dates to July 6 and differs from the current Antigravity model list. Usage evidence is scoped to the directories inspected; its age does not prove that no AI work occurred elsewhere. + +Historical planning records under .planning are local and gitignored. They were read as historical evidence, not treated as current instructions or proof that their claims remain true. Findings from those records are summarized here so the plan does not depend on readers possessing them. + +## How DevSquad arrived here + +| Period | What shipped or was recorded | What it teaches | +| --- | --- | --- | +| February 13 | Claude-oriented delegation plugin, marketplace restructuring, hook fixes, capacity reporting, acceptance tracking, and estimated token savings | Installation and execution wiring were part of the product problem from the beginning. Counting suggestions is different from completing delegated work. | +| February 18–19 | Git-health helpers, Gemini-to-Codex skill generation, and a real sequential shell workflow runner with gates/checkpoints | These are substantive utilities, but the unit of delivery became scripts and templates rather than completed engineering outcomes. | +| February 19 verification | A completion summary described the workflow as working end-to-end after syntax, JSON, grep, and dry-run checks. A separate verification record still called for real execution; a milestone audit identified cross-component runtime breaks | Structural verification was promoted into a stronger claim than the evidence supported. Future milestones need a recorded real run and its accepted artifact. | +| February 20–25 | Version renumbering, packaging corrections, and repeated validator/audit fixes through 0.3.0 | A source checkout working correctly does not establish that an installed plugin works in another project. | +| April 16 | 0.4.0 repaired new-project hook registration, paths, and noninteractive wrapper behavior | Fresh installation and ordinary project use must be acceptance tests, not post-release discoveries. | +| July 5–7 | Holdout experiment, Antigravity transition, Grok integration, common adapter, model catalog, expanded tests, and accepted ADR-001 | This is the strongest reusable engineering foundation. It also documents deployment drift, duplicated hooks, sparse observed delegation, and the need for a shared core. | +| July 7 onward | 0.10.0 maintenance release followed by README, growth, and motion work; August's main change is README-only | The accepted core/manifest/dispatch plan did not reach implementation. The next milestone should be smaller and centered on use. | + +Supporting commits include [initial implementation](https://github.com/joshidikshant/devsquad/commit/ac797bb), [February milestone](https://github.com/joshidikshant/devsquad/commit/a480e5a), [April fixes](https://github.com/joshidikshant/devsquad/commit/b98558e), [shared adapter](https://github.com/joshidikshant/devsquad/commit/5e6b08b), [ADR acceptance](https://github.com/joshidikshant/devsquad/commit/3a96bef), and [0.10.0](https://github.com/joshidikshant/devsquad/commit/5afc2f0). + +The July 5 routing record reports four completed delegation usage records plus test artifacts across its examined history. ADR-001 uses a different scope and describes approximately one meaningful completion and zero accepted suggestions out of 80. These are dated historical observations, not a new September adoption count. They support demanding better completion evidence; they do not establish that users or models reject the team concept. See [routing history](../../ROUTING-CHANGELOG.md) and [ADR-001](../adr/ADR-001-contract-and-ledger-core.md). + +My interpretation: DevSquad has accumulated substantial integration knowledge, but the feedback loop from ordinary use to accepted outcomes is weaker than the planning and review loop. More architectural discussion alone will not close that gap. + +## Readiness against the engineering team objective + +| Requirement | Current state | Practical consequence | +| --- | --- | --- | +| Reliable provider execution | Three thin wrappers and a common adapter; incident tests exist, but process cleanup and semantic success need work | Useful foundation, not yet a reliable unattended execution boundary | +| Interchangeable team leadership | Claude-specific hooks and eight Claude Sonnet relay agents; no neutral job command or Claude worker adapter | Codex/Astra cannot use the same team interface symmetrically | +| Roles matched to capabilities | Coarse keyword routes, hardcoded hook choices, mixed model families behind Gemini role names | Actual tools and model diversity do not reliably determine assignment | +| Complete implementation ownership | Agents are framed around short drafts; workflow runner executes shell strings | A coding agent is underused when a task needs sustained implementation and validation | +| Shared work and handoffs | Project state and text responses; no durable task/result contract or dependency-aware handoff | Recovery and integration depend on the lead reconstructing context | +| Independent review and QA | No explicit reviewer lane or acceptance gate in the core | An exit code or plausible explanation can be mistaken for a successful outcome | +| Capacity allocation | Manual usage cache and advisory messages; fixed cooldowns; Grok omitted from capacity schema | Capacity reporting does not yet allocate jobs or maximize usable subscription work | +| Safe parallel engineering | Some session-scoped counters, but shared mutable state and whole-tree checkpoint behavior remain | Parallel writers can collide or lose accounting; isolation must precede concurrency | +| Quality and value measurement | Character counts, suggestion outcomes, and Claude-focused holdout reconciliation | There is no evidence yet that the proposed team improves accepted engineering output | + +## Immediate blockers + +1. **Model identity is unreliable.** With the inspected local catalog/configuration, Gemini researcher/developer frontier pins resolve to Claude Opus 4.6 through Antigravity. The algorithm ranks cross-family numeric version strings. Preserve model family intent and distinguish harness, model provider, requested model, observed model, available tools, and account pool. See [model-catalog.sh](../../plugin/lib/model-catalog.sh), especially resolve_model_tier. +2. **Hook routing bypasses configured routing.** Reading and WebSearch are assigned to Gemini and tests to Codex directly. Unify this policy with workflow/manual dispatch. The lead can specify a role and required capabilities explicitly, avoiding brittle natural-language classification without adding another routing model. See [pre-tool-use.sh](../../plugin/hooks/scripts/pre-tool-use.sh) and [routing.sh](../../plugin/lib/routing.sh). +3. **Directory context is incomplete.** The file whitelist omits TSX, JSX, and other common source types; it would omit all nine TSX files in this project's motion source directory. String-split file arguments also mishandle spaces. Use an explicit file manifest, honor ignore rules, and report omissions and size limits. See [gemini-wrapper.sh](../../plugin/lib/gemini-wrapper.sh). +4. **The portable watchdog can add its full timeout to successful captured calls.** This Mac has no timeout/gtimeout binary. An immediate fake CLI response took 2.025 seconds with a two-second bound because the watchdog's sleep retained the capture pipe. Fix process/descriptor cleanup and verify elapsed-time behavior, timeout termination, and cancellation. See [adapter.sh](../../plugin/lib/adapter.sh). +5. **Process success is not task success.** Output may be empty, a tool may be denied, tests may never execute, or the requested artifact may not exist despite exit zero. Normalize execution status separately from acceptance, preserve provider events where supported, and require verifiable artifacts/checks. +6. **State and Git checkpoints need ownership.** JSON array rewrites can lose concurrent updates. Project-wide pending suggestions and workflow state mix sessions. Checkpoints stage all files and suppress commit failure. Use isolated run state, serialized event writes, scoped worktrees, and truthful checkpoint outcomes. See [usage.sh](../../plugin/lib/usage.sh), [enforcement.sh](../../plugin/lib/enforcement.sh), and [lib-workflow.sh](../../plugin/skills/workflow-orchestration/scripts/lib-workflow.sh). +7. **Installed-version behavior needs a health check.** Grok needs reauthentication. Local Antigravity/Grok versions must be checked against the features used, rather than inheriting assumptions from current documentation. Local duplicate DevSquad hook registration is presently absent; the installer/onboarding registration paths still need an idempotence test so this historical defect is not reintroduced. + +These blockers justify targeted fixes, not a wholesale rewrite. The exact affected behavior should have a regression test before being labeled repaired. + +## What “usable” must mean + +The first release should let the user state a bounded engineering objective from Codex or Claude, then receive a tested patch and an independent review without manually copying prompts between products. A provider becoming limited or unavailable must preserve the work, select a suitable alternative when available, and explain the resulting state. A restart must resume or truthfully report the interrupted task. + +Minimum observable experience: + +1. Inspect the available team once: runtime versions, authentication state, verified capabilities, and capacity freshness. +2. Give the lead an issue, a repository, and acceptance criteria in ordinary language. +3. See a compact plan with accountable roles; only independent work runs concurrently. +4. Receive progress when a result, failure, handoff, or decision matters. +5. Receive the patch, reviewer findings/dispositions, checks executed, unresolved limitations, and a compact capacity receipt. +6. Continue that task from the same saved artifacts if execution stops or leadership changes. + +The user should not have to choose every model, prepare workflow JSON for each task, re-explain the repository after every handoff, or repeatedly enter usage percentages. Natural-language interaction belongs to the active lead; the core should execute explicit structured jobs beneath it. + +The first workflow should be one real issue to implementation, independent review, correction, and test verification. Use two model families initially; add Grok or Gemini specializations when the issue benefits from them. A mandatory four-provider chain is not a requirement. A passing dry-run, generated skill scaffold, or impressive research report is not this milestone. + +## Minimum architecture + +Keep the runtime in the installed plugin package. Expose the existing wrappers through a small executable interface; thin Claude and Codex integrations call the same core. Existing native harnesses continue to own their agent loops, credentials, tool execution, and provider sessions. DevSquad owns job assignment, state, policy, artifacts, acceptance, and capacity accounting. + +Proposed interfaces, not current commands: squad doctor; squad invoke; squad run; squad status; squad resume; squad report. A thin Codex skill can use the CLI first. MCP is an optional later transport when actual host integration needs it, not a prerequisite. + +Four small contracts are sufficient to start: + +| Contract | Essential fields | +| --- | --- | +| Task | task/run ID, objective, repository and base revision, role, required capabilities, input artifacts, acceptance criteria, allowed operations, dependencies, deadline/budget constraints | +| Adapter capability | harness and installed version, model provider/family, requested/observed model, tool capability, verification status/date, auth route, account pool, output format, permission profile | +| Result/handoff | task/attempt ID, execution status, provider session ID if available, model/tool evidence, worktree and patch/artifact references, checks and results, reviewer findings, remaining work, source evidence when relevant, duration and measured usage | +| Capacity observation | account pool, applicable limit windows, remaining/used value when available, reset time when known, observed time, source, freshness, confidence, unavailable reason | + +Separate execution states from acceptance states. An attempt can exit successfully while its task remains unverified or needs revision. Distinguish failed, interrupted, waiting for quota, blocked capability, and awaiting a user decision. Do not turn every failure into a generic text response. + +Use one worktree and one writer at a time per implementation task, with a read-only reviewer over a recorded revision. Additional independent tasks can get their own worktrees. A handoff includes the base revision, current diff, relevant decisions, completed checks, outstanding failures, and next action; it does not pretend that hidden model context transfers between providers. + +Store per-run artifacts and events. A JSONL file is not automatically a concurrency solution: choose a serialized writer or explicit locking, stable event IDs, and crash-safe updates. Derived dashboards or reports can be rebuilt from those records. + +## Subscription capacity is a scheduling input + +Current capacity reporting asks users for quartile ranges and stores their midpoints. The schema omits Grok; recommendation logic does not use Codex's weekly value to choose its displayed zone and does not connect to job routing. Missing values can appear as zero usage. This is a reporting aid, not an allocator. See [capacity command](../../plugin/commands/capacity.md) and [usage.sh](../../plugin/lib/usage.sh). + +The next scheduler should: + +- Establish capability and minimum quality eligibility first, then consider fresh availability, relevant limit windows, resets, latency, and the role preference order. +- Group executors by their actual account/limit pool. Codex app and CLI are not two budgets merely because they are two interfaces. Model identity also does not establish that two runtimes share a pool. +- Represent unknown capacity as unknown. Distinguish subscription limits, model context occupancy, per-call tokens, temporary cooldowns, and paid API spending. +- Prefer supported automatic observations. Accept a timestamped manual observation when necessary, with honest precision; do not require user input before every dispatch. +- Track account-level capacity across projects, while keeping task state per run. Coordinate concurrent launches and use conservative concurrency caps when remaining capacity cannot be measured precisely. +- Preserve capacity for the lead and final review when the user needs them. Fill independent work with other capable providers where useful; do not spend quotas simply to achieve utilization. +- On rate limit, checkpoint before a compatible handoff; on authentication failure, mark the adapter unavailable until repaired. If a required unique capability has no alternative, report that limitation rather than silently substituting an unsupported answer. +- Keep subscription execution and separately billed API paths explicit. Adding a video API connector should not silently change the payment route of ordinary work. + +The objective is accepted engineering work per available subscription window, under a quality floor. Neither raw token savings nor equal use of every provider captures that objective. + +## Five delivery gates + +These are acceptance gates, not five large releases or calendar promises. Gates 1–3 can ship together as one narrow vertical slice using two already-working harnesses: repair the execution/context defects that affect that task, add only its required invocation/results boundary, and finish the task. A read-only branch-review command can provide utility even earlier. Full registry migration, all-provider support, and a new Claude worker adapter must not block that first loop; add the Claude worker when the chosen team needs it. Complete the wider provider matrix, installation hardening, and CI before the daily-use release. Gate 3 provides an early usable engineering loop; Gate 4 delivers the combined team-plus-capacity proposition. Gate 5 determines whether it is ready to become the default daily workflow. + +| Gate | Deliverable | Required acceptance evidence | +| --- | --- | --- | +| 1. Trustworthy execution | Repair execution/context/model issues exercised by the first two harnesses; report their health and use explicit role permissions | Existing suite passes; targeted regressions cover the selected path's elapsed time, cancellation, source manifest, model identity, and denied/empty output; supported-version smoke runs are recorded | +| 2. Shared jobs and results | Package a neutral executable around the needed adapters, with minimal manifests, structured task/results, and durable per-run artifacts | The same bounded job is callable from Claude and Codex; actual/unknown model identity is reported honestly; failure retains artifacts; the packaged interface works outside the DevSquad checkout | +| 3. Complete engineering loop | One issue to plan, isolated implementation, different-model review, bounded revision, and executed checks | A real patch satisfies its predefined criteria; review findings are resolved or explicitly dispositioned; failed prerequisites block dependents; no manual prompt ferrying is required | +| 4. Capacity and recovery | Pool-aware eligibility, fresh/unknown capacity, bounded fallback, checkpointed handoffs, resume, and controlled concurrency | Simulated rate limit and authentication failure are distinguished; a handoff preserves existing edits and checks; a shared pool is not double-counted; required capabilities survive fallback; interrupted tasks resume without duplicate application | +| 5. Daily-use validation and release | Representative task trials, doctor/CI coverage for all advertised adapters, coherent install/version/release procedure, and user-facing evidence | Clean-install smoke verifies one hook registration; advertised provider paths are checked; pilot records quality, completion, rework, intervention, time, and capacity evidence; runtime/package versions match; limitations are published with a tagged release | + +Use isolated fake providers for failure-injection tests. Live smoke calls establish what the real installed providers can execute; they should not deliberately exhaust quotas or invalidate working credentials. Repeated real use is necessary to establish reliability beyond either kind of test. + +The shortest useful proof is a modest bug fix or feature in an existing repository. Avoid using DevSquad's own skill generator as the only proof: that narrows evaluation to scaffolding, mixes generated content with plugin internals, and fails to exercise ordinary engineering ownership. + +## ADR-001: preserve the foundation, revise the product assumptions + +Preserve its Bash-compatible runtime, executable JSON boundary, in-plugin packaging, thin adapters, deterministic policy, ledger, and evidence before learned routing. Existing callers should remain compatible while the new path is tested. + +Propose a short follow-up ADR covering these changes rather than editing the accepted record to imply past agreement: + +1. Both Claude and Codex may lead; either may serve as an external worker/reviewer. +2. The first product workflow is engineering delivery and independent verification. GrowthSquad and a compulsory three-vendor demo move later. +3. Task acceptance and subscription utility become first-class outcomes. D1 remains an experiment about Claude delegation behavior, not the survival criterion for the whole team product. +4. Fallback preserves capability requirements. Unconditional terminal self-answering is not acceptable when the task requires evidence or operations the lead cannot perform. +5. Role-specific permissions replace blanket approval-bypass defaults; scoped acceptance policies should respect the user's already-authorized work. +6. Model choice uses identity, actual capabilities, and evidence rather than cross-family version sorting or a fixed historical latency floor. + +Do not pre-approve a language rewrite, licensing change, separate repository, marketplace, hosted service, universal provider registry, learned model router, or parallel multi-writer system. Add each only when observed use creates a specific need. Role templates can start small; a generator is not required for the first team loop. + +## How to test the combination thesis + +Start with roughly 10–15 representative tasks across bug fixes, small features, refactors, test improvements, and an occasional task requiring external evidence. This is a usability pilot, not a statistically conclusive benchmark. + +Compare DevSquad with the strongest single-agent workflow the user already uses. Keep task scope, starting revision, acceptance criteria, and available evidence comparable. Review resulting patches/reports without provider labels where practical. Judge criteria established before seeing the outputs; account for repeated-task learning when interpreting results. + +Record accepted completion, defects/omissions, human correction and intervention, time to a verified result, work preserved after interruption, and available quota observations. Capture provider usage when exposed, otherwise mark it unknown. Do not infer consumed subscription percentage directly from character counts. + +Test three separate sources of benefit: complementary tool access, independent design/review judgment, and continuity when one provider is constrained. A failed token-saving experiment does not refute all three; conversely, using more models does not prove any of them. + +Release decisions should name task classes where the team helps, where the direct single-agent path is preferable, and where the evidence is insufficient. Keep that fast direct path as part of the product. If review repeatedly adds no useful findings, reduce that lane for the relevant task class. If a provider contributes unique evidence or preserves progress under limits, record that value explicitly. + +## Provider facts relevant to the implementation + +Current official documentation supports native headless execution for all four families of harness. Exact flags, event dialects, permissions, and telemetry vary by installed version; the adapter must verify that contract rather than assume interchangeability. [Claude headless](https://code.claude.com/docs/en/headless), [Codex non-interactive mode](https://learn.chatgpt.com/docs/non-interactive-mode), [Grok Build headless](https://docs.x.ai/build/cli/headless-scripting), [Antigravity headless](https://antigravity.google/docs/cli/headless/). + +Grok Build documents X search. Gemini video understanding through its API supports YouTube URLs, but that does not establish the same path in the installed Antigravity CLI. Record separate web, X, transcript, and video capabilities. [Grok changelog](https://x.ai/build/changelog), [Gemini video input](https://ai.google.dev/gemini-api/docs/video-understanding). + +The repository's blanket statement that Gemini CLI was decommissioned is too broad: Google's June transition affected individual free/AI Pro/Ultra access, while enterprise and paid API access remain supported. Correct the account-specific wording when updating setup documentation. [Google transition announcement](https://developers.googleblog.com/en/an-important-update-transitioning-gemini-cli-to-antigravity-cli/). + +## Recommended immediate next implementation + +Begin with Gate 1 against an isolated fixture repository, then the shared job/result interface and one issue-to-reviewed-patch workflow. Seed the new tracked delivery record with the five gates above and their evidence links. Keep February's local planning records as history, not the active completion dashboard. Make installed-package verification and one real accepted run part of every milestone that claims end-to-end usability. + +The next proof should be a user task delivered by the team with less coordination burden, preserved work during a provider handoff, and a result that passes independent checks. That is the path from the existing delegation plugin to a usable engineering team. From d524b15b6ae512880a73df7c8efe1daea7ca1d6e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 6 Sep 2026 12:25:36 +0530 Subject: [PATCH 002/197] docs: define shared engineering team architecture and execution plan --- docs/ARCHITECTURE.md | 8 + ...02-surface-independent-engineering-team.md | 159 +++++++++++++++ docs/plans/engineering-team/CONTRACTS.md | 181 ++++++++++++++++++ docs/plans/engineering-team/IMPLEMENTATION.md | 138 +++++++++++++ docs/plans/engineering-team/START-HERE.md | 63 ++++++ docs/plans/engineering-team/backlog.json | 76 ++++++++ .../plans/engineering-team/examples/README.md | 10 + .../examples/branch-review.json | 45 +++++ .../examples/issue-delivery.json | 45 +++++ 9 files changed, 725 insertions(+) create mode 100644 docs/adr/ADR-002-surface-independent-engineering-team.md create mode 100644 docs/plans/engineering-team/CONTRACTS.md create mode 100644 docs/plans/engineering-team/IMPLEMENTATION.md create mode 100644 docs/plans/engineering-team/START-HERE.md create mode 100644 docs/plans/engineering-team/backlog.json create mode 100644 docs/plans/engineering-team/examples/README.md create mode 100644 docs/plans/engineering-team/examples/branch-review.json create mode 100644 docs/plans/engineering-team/examples/issue-delivery.json diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 4448f30..ca84c5c 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -2,6 +2,14 @@ One page for future maintainers (including future Claude sessions). +> **September 2026:** The diagram below describes the legacy Claude plugin; +> provider names, deployment details and measurements are historical. The +> [current assessment](audits/2026-09-06-engineering-team-assessment.md) records +> verified gaps. The proposed shared runner is specified in +> [ADR-002](adr/ADR-002-surface-independent-engineering-team.md), with a +> [coding-agent execution packet](plans/engineering-team/START-HERE.md). +> That new runtime is planned, not implemented. + ## End-to-end flow ```mermaid diff --git a/docs/adr/ADR-002-surface-independent-engineering-team.md b/docs/adr/ADR-002-surface-independent-engineering-team.md new file mode 100644 index 0000000..2216801 --- /dev/null +++ b/docs/adr/ADR-002-surface-independent-engineering-team.md @@ -0,0 +1,159 @@ +# ADR-002: One engineering team, accessible from every local surface + +- **Date:** 2026-09-06 +- **Status:** Implementation design prepared from Dikshant's brief. The architecture and backlog are ready for a coding agent; the runtime described here is **not implemented**. +- **Scope:** One user, one machine, multiple repositories, multiple local AI applications and subscription CLIs. +- **Basis:** [Repository and GitHub assessment](../audits/2026-09-06-engineering-team-assessment.md), source at `c7f5930`, and the user's subsequent requirements for model/effort selection, capacity, learning, documentation and surface independence. +- **Execution entry:** [START-HERE](../plans/engineering-team/START-HERE.md). + +## Decision in one picture + +```mermaid +flowchart TB + T[Terminal CLI] --> C[squad commands] + A[Codex App] --> M[Thin local MCP bridge] + B[Claude Code App / CLI] --> M + G[Antigravity IDE / CLI] --> M + X[Grok Build] --> M + M --> C + C --> R["Shared runner
Persisted runs and bounded workflows"] + R <--> S["SQLite events and state
Artifacts and handoffs"] + R --> P["Choose role + harness + model + effort
Capabilities first, capacity second"] + P --> W["Native CLI workers
Claude · Codex · Grok · Antigravity"] + W --> V[Independent review and checks] + V --> S + S --> L[Evidence → experiments → versioned policy] + L --> P +``` + +**The product is an engineering team with shared memory and verifiable work.** A provider's distinctive information access is one capability among coding, diagnosis, design, testing, review, tool use and synthesis. Roles stay stable; assignments change with evidence. + +Optimize accepted outcomes first, then reduce rework, latency and scarce allowance. Do not maximize the number of models involved, equalize provider use, or assume that more reasoning effort improves every task. + +## What remains and what changes + +[ADR-001](ADR-001-contract-and-ledger-core.md) remains the historical July decision. This design retains its packaged core inside `plugin/`, reusable adapters, bounded invocation, explicit capabilities, static initial routing and evidence requirements. For this proposed build it replaces these parts: + +| July design | September implementation decision | +|---|---| +| Bash-only coordinator | Python 3.11+ standard-library coordinator; preserve Bash 3.2 adapter compatibility | +| JSONL as primary ledger | Transactional SQLite event ledger and projections; JSONL is an export | +| Claude session always synthesizes | One explicitly selected lead: current host or a headless worker | +| Provider/role aliases and implicit self fallback | Exact execution profiles; qualified fallback or an explicit blocked state | +| Three-provider demonstration | Useful branch review first; two-harness delivery workflow next | +| D1 determines the platform's future | D1 evaluates hook enforcement only; engineering outcomes evaluate the team | +| Timing/volume threshold as learning prerequisite | Begin manual evidence-based improvements immediately; defer automatic learned routing | + +This does not retroactively mark new decisions as accepted in July. Old product descriptions and historical measurements are not evidence that the new runtime exists. + +## Runtime and packaging + +Add a small Python package inside `plugin/core/`. Its standard library provides JSON validation logic, SQLite, subprocess control and the CLI. Use the official Python MCP SDK as an **optional**, pinned dependency for the MCP bridge; ordinary CLI operations must work without it. Resolve and test the precise SDK version during implementation. + +```text +plugin/core/ + bin/squad stable command entry + pyproject.toml Python floor and optional MCP dependency + src/devsquad/ + cli.py, contracts.py one service API for CLI and MCP + store.py, supervisor.py transactions, ownership, process lifecycle + workflows.py, router.py two fixed workflows, profile selection + adapters.py, capacity.py provider bridge and account pools + learning.py, reports.py observations, evaluations, derived docs + mcp_server.py short tool calls; no separate business logic + schemas/ versioned public JSON contracts + adapters/ manifests and Bash bridge to existing wrappers + policies/ starter policy and reusable role prompts + integrations/ minimal instructions/config templates per host +test/core/ offline unit, process and integration tests +``` + +Use one local database, `~/.devsquad/runtime/state.sqlite3`, on local storage with WAL and schema migrations. Repository identity comes from the canonical Git common directory; worktrees share an identity. Store runs below `~/.devsquad/runtime/projects//runs//`. Runtime data stays outside Git. Versioned project policy lives in a new `devsquad/` directory, avoiding the existing ignored `.devsquad/` legacy state. + +The initial runtime needs no web server, queue service or permanent daemon. `squad start` persists a request, starts one detached supervisor per run and immediately returns its ID. Supervisors own native CLI subprocess groups. Any local surface can inspect, cancel or resume the same run. Closing an app or its MCP connection does not cancel a run. + +A standalone install exposes `~/.local/bin/squad` through a stable launcher to an immutable release directory under `~/.devsquad/releases/`. Plugin and standalone packages use the same core. Running jobs pin their release path and digest. An explicit development install may target the source checkout, with a dirty-source fingerprint. `doctor` reports installation drift. No runtime path should require an active Claude session or a versioned Claude cache path. + +## Surfaces are clients, not separate orchestrators + +| Surface | Integration | Boundary | +|---|---|---| +| Terminal | `squad` directly | Can use a supplied task and a headless lead | +| Codex desktop / CLI / IDE | Local stdio MCP | Local Codex surfaces share MCP configuration on the same host [1] | +| Claude Code desktop, Code tab / CLI | Local stdio MCP | Use shared user/project MCP configuration; local Code sessions have the relevant CLI integration [2] | +| Antigravity IDE / CLI | Local stdio MCP | Verify installed version against documented settings; configure one DevSquad server [3] | +| Grok Build | Local stdio MCP | Detect inherited Claude/project registrations before adding another [4] | + +V1 supports these **local** surfaces on the same machine. Cloud sessions need a later authenticated remote transport; a local stdio server does not make the Mac remotely reachable. Private chat history, hidden reasoning and host-specific tools do not transfer. Task specifications, files, results, decisions and handoff packets do. + +MCP exposes ordinary short `start/status/events/result/cancel/resume/handoff` operations. Do not depend on every host supporting MCP's optional long-running-task extension. Host-specific instructions explain when to use these operations; they must not contain their own router or provider matrix. + +## One lead, bounded workers + +The host supplies a structured task with acceptance criteria. In `lead.mode=host`, the current app is the lead and the runner executes a fixed workflow. When a decision is needed, the run becomes `awaiting_host` with a saved packet. Another surface can claim that handoff using a fenced ownership token. + +In `lead.mode=headless`, a selected CLI profile performs the same synthesis/disposition step. The core does not start an additional planner. V1 accepts structured tasks and two fixed workflow templates; natural-language planning and arbitrary workflow DAGs can come later. + +```mermaid +flowchart LR + I[Task + acceptance criteria] --> E[Implement in isolated worktree] + E --> R["Different-model review
Bound to patch hash"] + R --> Q[Checks on that revision] + Q --> D{Lead disposition} + D -->|bounded repair| E + D -->|criteria met| O[Verified result + patch + receipt] + D -->|unresolved| B[Blocked / failed with evidence] +``` + +Start with the smaller `branch-review` template: snapshot → review → checks → disposition → receipt. Add `issue-delivery` after this is useful. Repository changes occur in an isolated worktree with one writer. Reviewers have verified read-only permissions and a frozen revision. Checks use trusted argv arrays. Patch changes invalidate prior review/test acceptance. V1 returns work for integration; it does not merge, push, deploy or publish automatically. + +Preserve the existing shell wrapper API and four error prefixes. Extract shared argument-building/classification helpers for a new bridge. In new runs Python owns timeouts, cancellation, draining and reaping; the legacy wrapper keeps its separately repaired bounded invocation path. Do not nest two competing watchdogs or rewrite provider behavior in a second place. + +## Select configurations, measure outcomes + +An execution profile is `(harness, harness version, model family, exact model, effort, tools, permissions, account pool)`. Antigravity may expose several model families: the harness name alone does not identify the model. Every attempt records requested settings, observed settings and verification confidence. Native X, Google Search or YouTube access must be verified for the **selected harness/model/tool combination**; model branding or training history is insufficient. + +Routing starts with a small versioned preference list per role/task class. Filter for capability, permission, quality eligibility and verified model/effort support. Then consider every applicable quota window, cooldown, concurrency, deadline and latency. If no eligible profile is available, block with a reason. Never silently relax required capabilities or switch to paid API usage. + +Account pools span applications and repositories where the underlying allowance is shared. Provider observations have sources and expiry times; unknown allowance is unknown. DevSquad's concurrency reservations do not reserve quota with a provider. Spend estimates, token counts, characters and subscription allowance are distinct measurements. + +## Learning and documentation are part of completion + +```mermaid +flowchart LR + A["Every attempt
Settings, outputs, failures"] --> B["Task verdict
Checks, review, lead repairs"] + B --> C["Comparable evidence
Later user corrections included"] + C --> D["One hypothesis
Small budgeted experiment"] + D --> E{Quality and capacity evidence} + E -->|supported| F[Versioned policy change] + E -->|inconclusive| G[Keep current policy] + F --> H[Monitor drift / regressions] + H --> C +``` + +Use two loops: bounded corrections inside the current task, and deliberate policy improvement across tasks. Initial profile preferences are hypotheses, not a provider leaderboard. Retain failed attempts, fallbacks and lead rework; a repaired final success must not become an unqualified success for the original worker. Hold out evaluation tasks and avoid tuning and testing on the same answer. + +Runtime records generate a receipt at every terminal state and a handoff packet when waiting. Curated `devsquad/learning/` records track `observed → hypothesis → tested → adopted/rejected → revalidate`. Policy changes link evaluations and rollback versions. Update these artifacts as work completes; a later scheduled summary can be added when requested, but no background automation is installed by this plan. + +## Delivery gates + +| Gate | User-visible result | +|---|---| +| M1–M2 | Reliable adapter bridge and recoverable local run; fake process tests prove the lifecycle | +| M3 | A real branch review from terminal produces an inspectable receipt | +| M4 | Start in one local app, inspect/finish in another, using the same run ID | +| M5 | Bounded issue → implementation → different-model review → checks | +| M6 | Account-pool-aware scheduling and evidence-backed profile comparisons | +| M7 | Fresh installation, documentation and real smoke receipts for all requested surfaces | + +The detailed [contracts](../plans/engineering-team/CONTRACTS.md) and [work packages](../plans/engineering-team/IMPLEMENTATION.md) are normative for the build. Their gates replace claims based only on dry runs or syntax checks. + +## Source notes + +Product integration documentation checked on 2026-09-06; recheck at installation because paths and host versions change. + +1. [Codex MCP configuration](https://learn.chatgpt.com/docs/extend/mcp?surface=cli). +2. [Claude Code desktop](https://code.claude.com/docs/en/desktop). +3. [Antigravity MCP](https://antigravity.google/docs/mcp) and [CLI MCP](https://antigravity.google/docs/cli/mcp/). +4. [Grok Build MCP servers](https://docs.x.ai/build/features/mcp-servers). +5. [Official MCP Python SDK](https://github.com/modelcontextprotocol/python-sdk) and [optional MCP Tasks](https://modelcontextprotocol.io/extensions/tasks/overview). diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md new file mode 100644 index 0000000..d373c24 --- /dev/null +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -0,0 +1,181 @@ +# Engineering-team contracts v1 + +**Design contract, not an existing CLI reference.** These contracts implement [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md). M1 creates schemas; later milestones add the operations below without changing their meanings. Examples are deliberately fictional fixtures. + +## 1. Public operations + +All machine responses use `{schema_version: 1, ok: boolean, data: object|null, error: object|null}`. Exactly one of `data` and `error` is non-null. Errors contain `code`, `message`, `retryable` and structured `details`. JSON output contains no progress prose; stderr is for diagnostics. MCP invokes the same application functions as the CLI, without shelling out to string commands. + +| CLI | MCP tool | Effect | +|---|---|---| +| `squad doctor --json` | `squad_doctor` | Versions, capabilities, permissions, registration/install drift; metadata only | +| `squad start --task-file FILE --idempotency-key KEY [--supersedes-run RUN] --json` | `squad_start` | Validate, snapshot and persist; return run ID promptly | +| `squad status RUN --json` | `squad_status` | State, version, active steps, blockers, usage observations and next action | +| `squad events RUN --after CURSOR --limit N --json` | `squad_events` | Bounded event page and next cursor; no provider reasoning stream | +| `squad result RUN --json` | `squad_result` | Receipt and artifact references; `ready:false` before terminal state | +| `squad cancel RUN --json` | `squad_cancel` | Persist cancellation intent; return before termination finishes | +| `squad resume RUN [--recovery-file FILE] --json` | `squad_resume` | Reconcile ownership and effects; resume only if safe | +| `squad handoff claim RUN --expected-version N --owner OWNER [--claim-file FILE] --json` | `squad_handoff_claim` | Obtain or renew fenced host claim and input packet | +| `squad handoff complete RUN --claim-file FILE --decision-file FILE --json` | `squad_handoff_complete` | Submit disposition with current claim and evidence | + +MCP accepts parsed task/recovery/decision/claim objects instead of CLI file arguments, and optional `supersedes_run_id` on start. Bind request origin to the calling integration where possible; a user-supplied host label is provenance, not an authentication boundary. Local server operates as the same OS user. No network listener in v1. + +The MCP server entry is `squad mcp serve` over stdio. Its stdout is exclusively MCP transport traffic; logs go to stderr. Missing optional MCP dependencies produce actionable install guidance without preventing ordinary CLI use. + +`start --wait` is a CLI convenience using the same saved run. Ctrl-C exits observation without cancelling; print the run ID and explicit cancel command. Non-waiting accepted operations exit 0; invalid input exits 64; ownership/version conflict exits 75; runtime/internal failure exits 1. `--wait` exits 0 for `succeeded`, 2 for `blocked`/`awaiting_host`, 3 for `failed`, 4 for `cancelled`. `doctor` exits 1 when required readiness checks fail. A successfully retrieved failed run still gives `status` exit 0. + +Later M6 CLI additions: `capacity observe --file FILE`, `outcome add RUN --file FILE`, `report --project PATH`, `learn propose --project PATH`, `policy evaluate --experiment FILE`. These use the same envelope. Read-only reporting never modifies the active policy. + +Core operations are idempotent where specified, not “exactly once” execution of external effects. Reusing `(project_id, idempotency_key)` with an identical canonical request returns the original run; a different request hash returns `CONFLICT`. Retransmitted cancellations do not spawn another cleanup process. A repeated successful handoff completion returns its existing result if the submission hash matches; conflicting late completion is rejected and retained as an audit event. + +Hash the canonical **submitted** request, including an optional predecessor ID, before resolving moving refs or loading mutable policy contents. After structural validation and project identification, claim the key transactionally in a non-runnable `queued` record with `phase: preparing`. Only the owning preflight completes the immutable snapshots and enables execution. A replay returns that record without resolving refs again; concurrent preflights cannot overwrite it. A crash during preflight is recoverable under the same claim discipline. Syntax-invalid requests create no run; post-claim validation failures produce a failed receipt. `start` may return a preparing run while bounded preflight proceeds; it never waits for a provider call. + +## 2. Task and policy inputs + +The Task schema rejects unknown fields and checks finite bounds. It contains: + +| Field | Required meaning | +|---|---| +| `schema_version` | Integer `1` | +| `project.repo_path` | Existing absolute Git repository path, registered using canonical common directory | +| `project.base_ref`, `project.target_ref` | Resolve to immutable commit OIDs before launch; review compares base→target, delivery starts at target | +| `workflow` | `branch-review` or `issue-delivery`; no arbitrary shell/DAG step types | +| `goal`, `task_class` | Bounded objective and comparison class, e.g. `bugfix-python-small` | +| `acceptance[]` | Stable criterion `id`, concrete `description`, `evidence_kind` (`review`, `check`, `artifact`, `host`) | +| `checks[]` | `id`, `argv` string array, repository-relative `cwd`, `timeout_seconds`, `required_to_pass` | +| `scope` | Repository-relative `read_paths`, `write_paths`; non-empty writes only for delivery | +| `lead` | `mode: host` or `headless`; headless requires candidate profiles in policy | +| `routing` | `profiles_file`, `policy_file`; absolute or repository-relative trusted config files | +| `budget` | `wall_seconds` of active run execution, `max_provider_calls`, `max_revisions`, `max_fallbacks_per_step`; all finite non-negative integers, wall/calls positive | +| `origin` | `surface` label, optional `session_ref`; no authorization or remote invocation implied | + +Snapshot the task, resolved refs, config files and their hashes before enqueue. Do not resolve a moving branch again halfway through a run. V1 uses committed inputs only: if dirty files intersect declared scope, reject with an actionable `INPUT_INVALID` rather than silently omitting them. Explicit dirty-worktree snapshot support is deferred. Validate paths against traversal and symlink escape. Task checks and policies are execution authority: repo content and provider output cannot add commands or widen permissions. + +The CLI must print the resolved scope/check plan in validation output; an app lead should supply it from the user's actual task. No magic inference from a README's embedded instructions. Checks may legitimately have side effects; run them in the isolated workspace with the declared process policy. “Read-only reviewer” does not mean executing arbitrary repository scripts is safe to treat as read-only. + +`Policy` contains an `id`, `version`, workflow role candidate lists, per-task-class minimum quality status, `require_different_model_for_review`, optional `prefer_different_harness_for_review`, account-pool policy and experiment budget. V1 role names are `implementer`, `reviewer`, `lead`; checks are deterministic process steps, not model calls. Add a `researcher` role only when a workflow needs specific research artifacts. Don't make every task pay for research or a council. + +`Profile` contains: + +```text +id; harness; model_family; model_id; effort {value, transport}; +required_tools[]; permission_policy; account_pool_id; billing_mode; +quality_status; evidence_refs[] +``` + +`effort.transport` is `native`, `model_variant`, or `provider_default`. Values are native to that harness/model; never translate “high” into a numeric equivalent across vendors. Unsupported explicit effort fails validation. Provider default may be allowed, but its effective value remains unknown unless reported. `quality_status` is `unvalidated`, `trial`, `proven` or `suspended`, scoped to task class by policy evidence. Trial profiles are eligible only in explicitly permitted classes. Exact family and model IDs are mandatory for cross-model independence claims; inability to verify identity blocks that claim. + +Install-time discovery reports supported values and evidence (`documented`, `probed`, `unavailable`, `unknown`) with `checked_at`, CLI version and toolset hash. Selecting a known catalog entry verifies that it exists, not that the invocation used it: attempts retain separate `requested` and `observed` fields. Manually verified mappings may establish identity for a versioned harness; silent model fallback must never be labelled confirmed. + +## 3. Adapter execution boundary + +Retain `invoke_codex`, `invoke_gemini`, `invoke_grok` and their legacy stdout/exit/error contract. Add a Claude headless adapter. New execution uses a bridge around shared wrapper configuration and classification helpers: + +1. `prepare(request.json)` returns `LaunchSpec`: fixed manifest-selected executable, argv array, working directory, optional stdin artifact, allowlisted environment overrides, requested model/effort, parser version, permission evidence. The bridge may source bundled wrapper functions; it must not source a request-selected file. +2. Python launches that argv directly in a new process session/group, with no `shell=True` or `eval`, owns lifecycle and captures bounded stdout/stderr to files. +3. `classify(exit_code, stdout_file, stderr_file, timed_out)` returns the legacy-compatible error class or execution completion plus native model/usage/session metadata. Extract shared logic rather than implementing two independent classifiers. A Bash argv builder can send NUL-separated arguments to a bundled Python serializer; never interpolate JSON strings into shell code. + +The new bridge does not call `_adapter_invoke`'s watchdog or write legacy JSON usage arrays. Python writes new-run telemetry once. Existing sourced callers retain their old bookkeeping. Per-run model/effort/permission overrides must not edit shared configuration. Native CLI timeouts may act as an earlier provider limit, but Python remains the sole supervisor and cleanup owner. + +The four adapter error codes stay `RATE_LIMITED | AUTH_ERROR | TIMEOUT | CLI_ERROR`, with auth checked before rate. Additional **core** errors include `INPUT_INVALID`, `PROFILE_UNSUPPORTED`, `CAPABILITY_UNAVAILABLE`, `CONFLICT`, `RECOVERY_REQUIRED`, `BUDGET_EXHAUSTED`, `POLICY_DENIED`, `INTERNAL_ERROR`. Do not expand the legacy prefix enum to represent orchestration states. + +Execution completion is separate from deliverable validity and acceptance. Empty output, malformed required JSON, tool/permission denial or an authentication banner can invalidate a nominal exit-0 invocation. Save raw native output and the parser verdict. Prompt compliance alone does not prove a reviewer was read-only: require a verified native restriction or OS process policy; otherwise that role is unavailable. Do not use the wrappers' current blanket approval flags in new worker profiles. + +Use Git's tracked-file inventory and explicit task scope for context. Preserve filenames with spaces, TSX/JSX and other tracked extensions. Bound bytes and document exclusions; never silently truncate required evidence. Exclude runtime state, secrets and ignored files by default; an explicitly required ignored input needs an intentional input artifact. Large context should use native scoped filesystem access when supported rather than concatenating every file. + +Workers get `DEVSQUAD_WORKER=1`, run/attempt IDs and a delegation-depth guard. DevSquad's worker-facing MCP tools reject new team starts and workflow mutations, and legacy hooks honor the guard. Native authentication remains available through the provider's normal mechanism, but credentials and environment contents are not logged. Do not assume prompt text alone stops recursive delegation. + +## 4. Durable state, concurrency and recovery + +SQLite schema has `projects`, `runs`, `steps`, `attempts`, `events`, `artifacts`, `claims`, `pool_observations`, `pool_reservations`, `outcomes` and `schema_migrations`. Entity IDs are opaque UUIDs; event cursors, versions, fencing tokens and migration versions are integers. UTC timestamps accompany durations measured with a monotonic clock. Mutations increment a run `version`; the append-only event and affected projections commit together in one short transaction. File artifacts are atomically finalized and hashed before a transaction references them; crash-created unreferenced files are recoverable garbage, not valid results. + +```mermaid +stateDiagram-v2 + [*] --> queued + queued --> running + running --> awaiting_host + awaiting_host --> running + running --> blocked + blocked --> running: resume after reconciliation + running --> succeeded + running --> failed + queued --> cancelled + running --> cancelling + awaiting_host --> cancelled + blocked --> cancelling + cancelling --> cancelled: all children reaped or absence confirmed +``` + +Terminal states are `succeeded`, `failed`, `cancelled`; they are immutable. Retrying a terminal run uses `start --supersedes-run RUN` with a new idempotency key and a supplied task. Validate the predecessor is terminal and belongs to the same project; persist the link and resnapshot the new request. Terminal `resume` returns `CONFLICT` with this next action; it never creates a run. `blocked` has a typed reason and next action. A crash-recovery scan reconciles stuck `running` jobs. Event cursors are monotonically increasing database IDs, filtered per run (gaps are normal), with bounded pages. Preflight may transition `queued` directly to `failed`; cancellation of a queued preflight must fence its late snapshot completion. + +Enforce one active writer attempt per worktree and one supervisor claim per run transactionally. Store PID, process-group ID, process-start identity, heartbeat, attempt token and package digest. Lease expiry alone never licenses a replacement writer. On lost supervisor, inspect the child identity, pending writes and last artifact boundary. A live child remains owned; observation may reattach without relaunch. Ambiguous identity returns `RECOVERY_REQUIRED`; do not kill a reused PID. + +Cancel: persist intent → signal the verified process group → drain output → bounded grace (default five seconds) → kill if needed → reap/confirm absence → record terminal state. Total run and per-step deadlines include retries. Keep `cancelling` with diagnostic evidence if termination cannot be confirmed; no competing writer may start. Native detached remote effects outside this process group are a separate adapter limitation: unsupported tools cannot be advertised as safely cancellable. + +`budget.wall_seconds` counts elapsed active execution, including preflight, checks, retries and cleanup; parallel steps consume wall time once per run. It pauses in idle `queued` (excluding active preflight), `awaiting_host` and `blocked` only after all worker processes are stopped and ownership reconciled. Host thinking and quota waiting can therefore continue overnight without consuming execution allowance; record them separately as waiting/total elapsed time. Exhaustion stops work, performs bounded cleanup and ends `failed` with `BUDGET_EXHAUSTED`; cleanup may exceed the budget only to enforce safe termination. A lost heartbeat never pauses the clock while a child may still be running. + +Resume distinguishes (a) live work, (b) a verified resumable native session, and (c) interrupted work needing a new attempt. Native session IDs are only used if the installed adapter has a tested resume capability. Default interrupted writers need a recovery disposition referencing attempt ID, current workspace/patch hash, known effects and chosen checkpoint. Validate this evidence before creating the next attempt. A retry must not blindly repeat external effects. V1 worker policies disable publish/deploy and other irreversible remote actions. + +Host handoff claims use a bounded lease and increasing fencing token. `claim` needs the expected run version; `complete` needs the current token, handoff ID, submission ID/hash and evidence refs. An expired host can no longer advance the run. Save its late submission for audit without applying it. `accept`, `revise`, `reject` are the allowed dispositions. Acceptance cannot override a failed mandatory check, stale revision, exhausted correction budget or missing independent review. + +Default host lease is ten minutes. A competing claim while a live claim exists returns `CONFLICT`; no implicit takeover. The current owner renews with `claim` plus its claim object and current expected version before expiry. Renewal extends the same fencing token; after expiry a successful new claim increments the token, even for the same owner label. Claim responses contain handoff ID, token, expiry and run version. Cancellation invalidates claims. This lease governs permission to submit a decision, not the lifetime of the saved handoff. + +## 5. Workflows and acceptance + +| Template | Ordered steps | Completion | +|---|---|---| +| `branch-review` | Resolve base/target → frozen review workspace → reviewer → configured checks → lead disposition | Valid review artifact, evidence for each criterion, recorded check outcomes, accepted disposition | +| `issue-delivery` | Snapshot target → implementer → frozen candidate → different-model reviewer → checks → lead disposition | Candidate meets all criteria, independent review complete, mandatory checks pass | + +For review-only tasks a check may have `required_to_pass:false`: a failing test then becomes a finding in a successfully delivered review. Delivery's required checks must pass. Mark these policies before execution. A valid critical finding is a useful reviewer contribution, not automatically a failed reviewer attempt. + +For `issue-delivery`, `revise` loops back to implementation. For `branch-review`, it requests a corrected review of the same frozen diff; it never authorizes source edits. Both consume one revision allowance and require a reason. With no allowance remaining, a revise request ends the run failed with `BUDGET_EXHAUSTED` after cleanup. Each role may attempt bounded fallbacks; permission/capability restrictions carry across them. Nonrecoverable step failure blocks dependent steps. No continuing into acceptance after missing implementation or invalid review. A headless lead's invalid disposition is a failed attempt, not permission to invent success; changing lead mode requires a new explicitly supplied task/run in v1. + +Bind artifacts, findings, tests, dispositions and criteria to a candidate tree/patch hash and resolved baseline. Freeze an implementation candidate into a local run-owned commit before review; reject changes outside scope and capture untracked permitted outputs intentionally. Run checks in a separate candidate worktree so test-generated files cannot alter the reviewed candidate. If a check must change source, that creates a new candidate needing new review. A subsequent code change invalidates affected evidence and triggers review/checks again. Run results contain a patch/commit reference and integration instructions; the coordinator does not alter the user's current checkout. + +Every attempt receipt includes role, parent step, profile/policy/prompt/schema versions, requested/observed model/effort, tools/permissions, runtime version, input/output hashes, process verdict, artifact verdict, latency, usage by source and error. The run receipt also includes criteria results, all attempt IDs, fallbacks, revisions, lead repairs, final disposition and remaining limitations. Host work is recorded as externally observed with unknown usage where unmeasured; do not omit it or estimate it as zero. + +## 6. Capacity without false precision + +Model providers, execution harnesses and billing accounts are separate identities. `account_pool_id` is a user-configured opaque identifier; no credentials. A pool may contain multiple windows and model-family sublimits. Unknown mapping is explicitly unresolved; do not assume two apps give two independent allowances or that every model within an account shares one limit. + +Observation fields: `pool_id`, `window_id`, `applies_to`, `observed_at`, `expires_at`, `source`, `used`, `limit`, `unit`, `resets_at`, `confidence`. Unknown measurements are null. Sources distinguish native reported values, manual reports and estimates. Percent observations retain their native unit; do not convert characters or token estimates into subscription percentage. Validate bounds and clock skew. Ignore stale observations for hard capacity decisions, while showing their last-known values. + +Route order: + +1. Capability, permissions, model/effort support and task-class quality eligibility. +2. All applicable fresh quota windows, auth failure, known cooldown and local in-flight concurrency/reserve constraints. +3. Versioned candidate preference, deadline and measured latency; log exclusions and selection rationale. + +An explicit `unknown_capacity_policy` is `allow_bounded` or `block`; default `allow_bounded` permits one short trial/in-flight job per unresolved pool, with normal per-run budgets, and labels uncertainty. A fresh known exhausted window blocks all affected profiles. Reset timestamps permit a refresh/reconsideration; they do not prove fresh availability. Auth errors require observed repair; rate errors use provider retry/reset metadata or conservative recorded cooldown. Never loop until a provider happens to recover. + +Local reservations are concurrency/scheduling records, not provider quota guarantees. Other apps can consume the allowance during a run. An account-level usage delta is not assigned wholesale to one task. Record per-call usage only when the provider identifies it, and preserve uncertainty. Billing mode is `subscription` or `paid_api`; policy must explicitly permit paid API profiles before selection. A subscription's marginal cash price is not its opportunity cost. + +## 7. Learning, evaluation and ongoing docs + +Run completion writes local `receipt.json`, `receipt.md`, `events.jsonl` export and an artifact manifest, including failures/cancellation. Waiting writes `handoff.json` and `handoff.md`. Raw outputs stay local. Tracked distilled records use: + +```text +devsquad/ + profiles.json, policy.json versioned intended configuration + learning/observations/.md outcome + provenance + later correction + learning/experiments/.json question, comparison, budget, stop rule + learning/evaluations/.md cases, failures, uncertainty, verdict + learning/decisions/.md adopt/reject/no-change + rollback target + learning/policy-changelog.md links to evidence and policy diffs +``` + +`report` derives summaries from the database; `learn propose` writes draft distilled records for review. It does not change routing. Raw task content is not automatically committed or uploaded. Track redacted evidence with reproducible case IDs/hashes and an explicit `evidence_availability` value (`local`, `tracked_fixture`, `unavailable`); a local path alone is not portable proof. Policy promotion requires a reviewed, versioned change with evaluation links. During M6 use an ordinary Git diff/review for this, not another bespoke approval UI. + +Use this loop: + +1. Define task class, criteria and intended comparison before execution. +2. Record every attempt, including failure, fallback, reviewer contribution and lead repair. Attach later corrections/escaped defects via `outcome add`; append revisions to verdicts, never erase the original. +3. Propose one change: model, effort, prompt/context strategy, tool access or workflow. Compare like tasks and keep the remaining settings controlled or explicitly record confounders. +4. Run a small predeclared evaluation with held-out cases, budget and stopping rule. Avoid routing only hard tasks to one model and then treating unadjusted averages as model quality. +5. Promote only with supported quality evidence, acceptable rework/latency/capacity tradeoff and a rollback target; otherwise preserve the policy and record no-change. +6. Revalidate affected profiles after model/harness/prompt/tool/policy drift or escaped defects. Do not discard unrelated evidence or automatically promote a newly released model. + +Default experiment budget is disabled until explicitly configured, then at most 10% of eligible runs with a hard call/time cap. Most work uses the current proven policy. V1 uses human-governed static preferences, not exhaustive permutations, an automatic bandit or foundation-model fine-tuning. Report sample sizes and missingness; tiny samples justify hypotheses, not provider rankings. + +Measure acceptance and critical defects first; also show retries, lead rework, elapsed time, measured usage by pool, blocked time and unmeasured overhead. Final task success and original worker quality are distinct. Pair deterministic checks with review and human correction; a model judging itself is not sufficient evidence. diff --git a/docs/plans/engineering-team/IMPLEMENTATION.md b/docs/plans/engineering-team/IMPLEMENTATION.md new file mode 100644 index 0000000..1cafc6f --- /dev/null +++ b/docs/plans/engineering-team/IMPLEMENTATION.md @@ -0,0 +1,138 @@ +# Implementation work packages + +**All milestones are pending.** This is the execution sequence for [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md) and [contracts v1](CONTRACTS.md). The existing 177 offline assertions passed during architecture preparation; they validate legacy behavior, not the proposed runtime. + +## Sequence and stopping points + +```mermaid +flowchart LR + M1[M1: Invocation contract] --> M2[M2: Durable runner] + M2 --> M3["M3: Branch review
First useful product"] + M3 --> M4[M4: Local app access] + M3 --> M5[M5: Code delivery] + M4 --> M6[M6: Capacity and learning] + M5 --> M6 + M6 --> M7[M7: Install and prove] +``` + +Implement sequentially through M3. M4 and M5 may proceed in parallel after agreeing on the frozen M3 public service API; assign separate files. M6 integrates their evidence. Each milestone can contain small reviewable commits. A milestone is complete only when its required gate has evidence; an unavailable provider or app leaves the relevant live gate open. + +## M1 — Make invocation truthful and reusable + +**Outcome:** A versioned profile selects a verified model/effort/permission combination, and the core can prepare/classify a call without owning a second watchdog. + +**Existing files:** `plugin/lib/adapter.sh`, `codex-wrapper.sh`, `gemini-wrapper.sh`, `grok-wrapper.sh`, `model-catalog.sh`; `test/test_wrapper_contract.sh`, `test/test_models.sh`. + +**New files:** `plugin/core/pyproject.toml`, `bin/squad`, `src/devsquad/{cli,contracts,adapters}.py`, `schemas/`, `adapters/{codex,antigravity,grok}/`; `test/core/test_contracts.py`, `test/core/test_adapters.py` and fake executables. + +Work in this order: + +1. Establish package/import layout, Python 3.11 floor, `squad --version`, initial `doctor`, strict v1 schemas and fixture loading. Record package/source fingerprints. +2. Extract reusable argv-building and classification helpers without changing sourced-wrapper signatures or error prefixes. Implement `prepare`/`classify` bridge; retain legacy telemetry only on legacy invocations. +3. Fix legacy portable-watchdog completion delay and descendant cleanup with timing/process assertions. Preserve Bash 3.2 and jq-absent legacy tests. Do not make legacy operation require installing Python. +4. Add per-call explicit model/effort/tool/permission settings; verify installed CLI mappings with help/documentation and bounded probes. Remove cross-family numeric tier ranking from new selection; correct legacy resolution with compatibility tests. Catalog drift produces unknown/revalidation rather than silent identity substitution. +5. Replace the Gemini extension whitelist and whitespace splitting with scoped, bounded context enumeration. Document when native file access replaces prompt concatenation. + +**Acceptance gate:** Existing suite passes. Fake immediate CLI returns promptly with a 2-second timeout (target under 1 second on normal local CI); a hanging CLI and descendant are gone by timeout plus grace. Fixtures cover empty exit-0, exit-0 auth banner, denied tool, malformed output, spaces/TSX inputs, explicit unsupported effort, unavailable capability and cross-family catalog entries. Overrides do not mutate global/project config. Doctor distinguishes supported, unverified and unavailable settings. One short read-only real adapter probe validates the chosen starting profile; save a redacted receipt and version facts. + +**Boundary:** No learned routing, MCP, worktree edits or universal model catalog. Manifests for unprobed harnesses remain visibly unverified. + +## M2 — Persist jobs and own their processes + +**Outcome:** Start/status/cancel/resume operate on the same saved job, including after the initiating shell exits. + +**New files:** `src/devsquad/{store,supervisor}.py`, database migrations, artifact storage helpers; `test/core/test_store.py`, `test/core/test_supervisor.py`. + +1. Implement SQLite ledger/projections, request hashing, schema migration and atomic artifacts under the machine-local runtime directory. Canonical Git common directory identifies a project; project state and shared account pools remain distinct. +2. Add start/status/events/result/cancel/resume service functions and CLI handlers. Persist the task before spawn; recover a crash between enqueue and spawn. Use one detached supervisor per run and fixed package snapshot. +3. Implement process-group ownership, bounded output capture, timeout/cancel/reaping, supervisor claims and a single active writer constraint. Native output goes to artifacts, not an unbounded in-memory buffer. +4. Implement recovery classification and host handoff storage/claim fencing. The public API needs these invariants before apps consume it. Expired heartbeat must not trigger blind relaunch. + +**Acceptance gate:** Two concurrent identical starts return one run; same key/different body conflicts. Event/projection mutations remain consistent under concurrent writers. Kill the supervisor while its child stays alive: resume launches no second writer. Reused/ambiguous PID is not signalled. Kill between artifact write and DB commit: no false completion. Repeated cancel is harmless; cancellation waits for confirmed child cleanup. Close the launching shell and inspect the job from another shell. Apply/test a database migration against a fixture of the preceding schema; refuse a newer unsupported schema. All tests use temporary runtime directories. + +**Boundary:** A fake-step harness exercises lifecycle; no claim yet that an engineering workflow works. No HTTP listener or permanent daemon. + +## M3 — Ship a useful branch review + +**Outcome:** `squad start` reviews a real committed diff and returns findings, check results and a receipt without modifying the user's checkout. + +**New files:** `src/devsquad/{workflows,router,reports}.py`, `policies/`, role templates, branch-review schema/fixtures; `test/core/test_review_workflow.py`. + +1. Implement profile/policy loading, version/hash snapshots and capability/permission/quality filters. Use a small static candidate list. Add basic shared-pool concurrency and typed unknown capacity now; richer observations arrive in M6. +2. Resolve commit refs and task scope; create a frozen review workspace. Reject intersecting dirty inputs rather than silently omitting them. +3. Implement reviewer → trusted checks → lead disposition, with `required_to_pass` semantics. Support host handoffs through CLI and a headless lead profile. Expose clear `awaiting_host` packets. +4. Produce receipt JSON/Markdown, events export and artifact manifest for every terminal outcome. Record unknown host usage honestly. + +**Acceptance gate:** An actual branch review returns actionable findings or a supported clean verdict, against recorded base/target hashes. A report-only failed check is included in a successful review; a required failed check prevents acceptance. A denied reviewer write or missing output cannot count as a valid review. A moving branch does not change the frozen run input. A second terminal claims a saved host handoff; a late completion from the first is fenced out. Verify original checkout/index/HEAD are unchanged. + +**First product stop:** At this point DevSquad is usable from terminal for a bounded job. Demonstrate it before adding more architecture. + +## M4 — Use that same run from local apps + +**Outcome:** Apps are interchangeable clients of the saved run. + +**New files:** `src/devsquad/mcp_server.py`, `integrations/{codex,claude-code,antigravity,grok}/`, optional MCP dependency/lock, generated tool-reference docs; `test/core/test_mcp.py`. + +1. Pin a currently supported official MCP Python SDK version and test it with the chosen Python floor and local runtime. Keep core CLI imports free of optional MCP dependencies. +2. Map the contract's operations directly to service functions. Paginate events, cap artifact previews and return IDs/paths/hashes. Start/cancel return promptly; app tool timeout is never the worker lifetime. +3. Add minimal host instructions: construct task/criteria, submit once with idempotency key, inspect status, claim a handoff, consume receipts. No host-specific router or copied provider matrix. +4. Create explicit setup templates for the documented local stdio integration. Detect existing/inherited registrations and stable executable resolution; preserve unrelated settings. Doctor reports what each installed app actually loads. +5. Enforce the worker recursion guard at MCP service and legacy hook boundaries. Worker sessions with inherited global MCP configuration must not create nested teams. + +**Acceptance gate:** MCP schema/conformance tests cover malformed requests and short tool timeouts. Start in terminal, inspect from Codex, complete a host handoff from Claude Code: the run ID, input hashes and ledger are identical. Close the MCP client and confirm the worker survives. Two hosts cannot both advance a handoff; stale host output is retained only as audit evidence. Prove one inherited duplicate server is detected and worker-origin mutation is rejected. Save installed-version receipts; config syntax alone is insufficient. + +**Boundary:** Antigravity and Grok configs can be prepared here; M7 requires their real local smoke receipts. No promise of hosted/cloud access. + +## M5 — Deliver a bounded engineering change + +**Outcome:** One model implements, another model reviews, checks verify the exact candidate, and the lead resolves the result. + +**Existing files:** Adapter manifests/bridge, workflow service, role prompts; new `plugin/lib/claude-wrapper.sh` and corresponding manifest if Claude headless is not yet available. Extend `test/core/test_adapters.py` and add `test/core/test_delivery_workflow.py`. + +1. Add the Claude headless adapter using the same conformance contract and verified permissions. Do not confuse the Claude Code host with a separately spawned Claude worker. +2. Implement isolated delivery worktree, one-writer enforcement, scope validation and local candidate commits. Return patch/commit artifacts without automatic merge/push. +3. Select a reviewer with a different verified model identity. Prefer a different harness when a qualified alternative exists; same-family or unknown identity must not masquerade as independent model review. +4. Run checks in a separate candidate worktree. Bind review, checks and disposition to candidate hash. Implement bounded revisions/fallbacks/deadlines and dependent-step failure handling. +5. Preserve all attempts and repairs in the result. Do not allow a failed mandatory check or unsupported reviewer restriction to be overridden by lead prose. + +**Acceptance gate:** A real bounded issue completes implementation → different-model review → correction if needed → checks → receipt using at least two harnesses. A fake reviewer catches a seeded defect and the corrected candidate reruns checks/review. Change the patch after a valid review: stale evidence cannot complete the run. Simulate rate-limit fallback without widening permissions. Kill a live implementation supervisor and prove no duplicate writer on resume. Verify no original-checkout changes or remote publication. + +**Boundary:** No arbitrary DAG, broad autonomous project implementation or automatic integration service. + +## M6 — Make capacity and improvement evidence useful + +**Outcome:** Scheduling respects shared limits, and completed work produces actionable, traceable policy proposals. + +**New files:** `src/devsquad/{capacity,learning}.py`, observation/experiment/outcome schemas, learning templates; `test/core/test_capacity.py`, `test/core/test_learning.py`. + +1. Implement pool mapping, applicable quota windows, TTL/source/confidence and local reservations. Use documented provider observations where available and timestamped manual values otherwise. Status displays unknown and stale values explicitly. +2. Add bounded fallback, native retry metadata and explicit paid-API eligibility. Snapshot every routing decision with exclusions and policy version. +3. Add final/late outcome records, role contribution, lead repair and evidence references; produce comparison reports with sample sizes and missingness. +4. Implement experiment specs and held-out evaluation fixtures. `learn propose` generates a draft hypothesis/evaluation/decision packet; promotion remains a reviewed Git policy change with a rollback target. +5. Generate receipts/handoffs on run transitions, and curated documentation on explicit report/proposal operations. Record drift and affected evidence; do not introduce an unrequested scheduled automation. + +**Acceptance gate:** Two projects sharing one pool obey a fresh exhausted weekly window despite available short-window capacity. Stale/unknown values never become zero; external usage changes do not get assigned to one worker. Concurrency reservations release only after ownership is reconciled. A paid API fallback is excluded unless allowed. A failed original attempt later repaired by another model produces final success without crediting the original as independently successful. A late escaped bug updates outcome history. A one-variable fixture experiment produces a traceable no-change or promotion proposal, with all failures and a rollback version; insufficient evidence leaves active policy unchanged. Re-run a held-out fixture after policy change and exercise rollback. + +**Boundary:** No automatic learned router. Experiment budgets default off; activate only through explicit versioned policy. + +## M7 — Package, migrate and prove every requested surface + +**Outcome:** A fresh install has one known runtime, reliable documentation and real smoke evidence. + +**Files:** New `scripts/install-core.sh`, packaging/release checks and install tests; update `install.sh`, `CONTRIBUTING.md`, `docs/ARCHITECTURE.md`, appropriate `plugin/commands/`/skills/hooks, release metadata and changelog when releasing. + +1. Package the core inside `plugin/`, with a stable standalone launcher and optional MCP environment. Preserve active run release pins. Installer is idempotent and reports source/plugin/standalone drift. +2. Keep legacy plugin mode available during migration. Reconcile hook registration only when duplicate evidence exists; September review found no current duplicate. New hooks call the shared route source after its tests pass and remain fast/network-free. No Python dependency imposed on legacy mode. +3. Create generated command/schema examples and concise install/operate/recover guides. Mark implemented vs deferred features; link completion claims to receipts. Keep historical ADR/audit statements dated. +4. Add offline CI for legacy and core suites (Bash 3.2/macOS compatibility and chosen Python floor/current version), package-content checks, and optional MCP tests. Live provider/app tests remain explicit bounded smoke runs. +5. Verify terminal CLI, Codex App, Claude Code App local Code tab, Antigravity local IDE/CLI and Grok Build against the same saved runtime. Check native capabilities/profile identity as used, not by brand inference. + +**Acceptance gate:** Fresh standalone install works without Claude installed; existing Claude plugin install contains `plugin/core` contents correctly. Reinstall creates no duplicate hook/server registration and a release update does not break a running job. Record actual start/observe/handoff-or-cancel receipts from every listed surface. Complete one end-to-end delivery with a different-model reviewer after installation. Documentation commands run as written. If an installed host cannot support an operation, retain that item as blocked with exact evidence instead of declaring universal support. + +## Evidence and ongoing status + +Update [backlog.json](backlog.json) as work proceeds. Allowed states: `pending`, `in_progress`, `blocked`, `complete`. A blocked item needs a precise reason and remaining independent work; completion requires non-empty evidence. Each evidence item records `kind`, `revision`, `command_or_action`, `outcome`, `artifact`, `recorded_at` and `availability`. Use an implementation revision already committed when writing a subsequent completion receipt; do not insert a circular “this commit's hash” placeholder. + +Store portable redacted milestone receipts under `docs/plans/engineering-team/evidence/`; reference private local run IDs separately. Never commit provider secrets or unreviewed raw task logs. Record measurements honestly: fake tests prove mechanics; live runs prove integration; neither alone proves a profile is generally better. + +Before each commit, run `bash test/run.sh`, relevant new tests, schema/example validation and `git diff --check`. Broaden testing only for changed behavior or unresolved concerns. Before ending, checkpoint the verified work and leave a clean tree under the repository's git-safety instructions. diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md new file mode 100644 index 0000000..eecabaa --- /dev/null +++ b/docs/plans/engineering-team/START-HERE.md @@ -0,0 +1,63 @@ +# DevSquad: coding-agent entry point + +**Build status: planned, not implemented.** This packet follows the local/GitHub review and Dikshant's September 6 brief. Start with M1; do not run another open-ended architecture exercise. + +> Build an AI engineering team that can be operated from terminal, Codex, Claude Code, Antigravity and Grok. Use each eligible model/effort/tool configuration where it produces the best verified outcome; account for shared subscription limits. Preserve work across surfaces, learn from attempts and maintain the evidence automatically. + +```mermaid +flowchart LR + A[Fix invocation] --> B[Persist and recover runs] + B --> C[Usable branch review] + C --> D[Same run from any local app] + C --> E[Implement → review → verify] + D --> F[Capacity + learning] + E --> F + F --> G[Install and prove the full workflow] +``` + +## Read and execute + +1. Read [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md) for boundaries and decisions. +2. Implement the [contracts](CONTRACTS.md), using the [examples](examples/branch-review.json) as fixtures, not live model configuration. +3. Work through [IMPLEMENTATION](IMPLEMENTATION.md), one milestone at a time. [backlog.json](backlog.json) is the completion record; all milestones initially have `status: pending` and empty evidence. +4. Consult the [assessment](../../audits/2026-09-06-engineering-team-assessment.md) for verified defects and history, and [ADR-001](../../adr/ADR-001-contract-and-ledger-core.md) for legacy constraints retained by ADR-002. + +## Copyable execution brief + +```text +Implement DevSquad's September engineering-team plan in this repository. +Read docs/plans/engineering-team/START-HERE.md and its contracts first. +Start at the earliest pending milestone whose dependencies are complete. +For the initial pass, implement M1 and pass its acceptance gate before M2. +Preserve existing Bash 3.2 wrapper callers and their four error prefixes. +Keep all distributable core files inside plugin/core; add no cloud service. +Use fake CLIs for development; real provider runs are bounded smoke tests. +Do not change global AI account settings or silently switch to paid APIs. +Before marking a milestone complete, record the revision, checks, outcome, +and a reproducible receipt in backlog.json and update the relevant docs. +Run bash test/run.sh before each commit as CONTRIBUTING.md requires. +Commit each verified milestone. Continue through ready work when requested; +report exact blockers rather than claiming unsupported integrations work. +``` + +The initial architecture is decided. Routine implementation choices need no renewed architecture approval. If a discovery changes a contract, document the small proposed change and its impact in an ADR amendment; preserve compatibility or version the contract. + +## Completion means evidence + +| Claim | Required proof | +|---|---| +| Adapter works | Fake-binary conformance, supported flag mapping, one bounded real smoke receipt | +| Recovery works | Crash with live child; resume never creates a second writer | +| Surface works | Start/status/handoff or cancel from that installed application | +| Delivery works | Exact patch, independent review, checks and disposition all linked | +| Router improved | Comparable cases, all attempts, failures and rework; versioned evaluation | + +Keep planned and implemented features visibly separate. Do not mark M4/M7 complete merely because a config file parses, or a host advertises MCP support. Do not rewrite `.planning/STATE.md`'s historical February “100%” into a claim about this build. + +## Scope guard + +First usable product: **a saved branch review**. Next: **one bounded code change reviewed by another model**. Two functioning harnesses are sufficient to prove the engineering workflow; M7 verifies access from every requested local surface. Do not force every provider into every run. + +Defer a dashboard, universal DAG builder, remote execution service, automatic model training/router, plugin marketplace, autonomous merges and scheduled documentation jobs. Existing plugin behavior remains available while the new runner is opt-in; switching hook suggestions to the new route source happens only after its gate passes. + +No new runtime, host registration, provider call, deployment or automation was created by this architecture packet. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json new file mode 100644 index 0000000..297f638 --- /dev/null +++ b/docs/plans/engineering-team/backlog.json @@ -0,0 +1,76 @@ +{ + "schema_version": 1, + "plan_id": "surface-independent-engineering-team", + "created_at": "2026-09-06", + "baseline_revision": "c7f5930", + "architecture": "../../adr/ADR-002-surface-independent-engineering-team.md", + "contracts": "CONTRACTS.md", + "implementation": "IMPLEMENTATION.md", + "status": "planned", + "next_milestone": "M1", + "milestones": [ + { + "id": "M1", + "title": "Truthful reusable adapter invocation", + "depends_on": [], + "status": "pending", + "acceptance_section": "M1 — Make invocation truthful and reusable", + "evidence": [], + "blocker": null + }, + { + "id": "M2", + "title": "Durable runs, cancellation and recovery", + "depends_on": ["M1"], + "status": "pending", + "acceptance_section": "M2 — Persist jobs and own their processes", + "evidence": [], + "blocker": null + }, + { + "id": "M3", + "title": "Usable branch review from terminal", + "depends_on": ["M2"], + "status": "pending", + "acceptance_section": "M3 — Ship a useful branch review", + "evidence": [], + "blocker": null + }, + { + "id": "M4", + "title": "Shared local MCP access and host handoffs", + "depends_on": ["M3"], + "status": "pending", + "acceptance_section": "M4 — Use that same run from local apps", + "evidence": [], + "blocker": null + }, + { + "id": "M5", + "title": "Bounded implementation with independent review", + "depends_on": ["M3"], + "status": "pending", + "acceptance_section": "M5 — Deliver a bounded engineering change", + "evidence": [], + "blocker": null + }, + { + "id": "M6", + "title": "Shared capacity and evidence-based improvement", + "depends_on": ["M4", "M5"], + "status": "pending", + "acceptance_section": "M6 — Make capacity and improvement evidence useful", + "evidence": [], + "blocker": null + }, + { + "id": "M7", + "title": "Installation, migration and all-surface proof", + "depends_on": ["M6"], + "status": "pending", + "acceptance_section": "M7 — Package, migrate and prove every requested surface", + "evidence": [], + "blocker": null + } + ] +} diff --git a/docs/plans/engineering-team/examples/README.md b/docs/plans/engineering-team/examples/README.md new file mode 100644 index 0000000..b048d9e --- /dev/null +++ b/docs/plans/engineering-team/examples/README.md @@ -0,0 +1,10 @@ +# Design fixtures + +These task files show the v1 shape. They are **not runnable against today's DevSquad**. Repository paths, refs, task classes and test commands refer to a future disposable fixture repository. M1 creates the schemas; M3/M5 create the corresponding repositories and executable tests. + +- [branch-review.json](branch-review.json): a host lead receives the review packet; a failing report-only check becomes a finding. +- [issue-delivery.json](issue-delivery.json): a headless lead resolves a bounded implementation/review workflow; required tests must pass. + +The fixture test harness must install a `devsquad/profiles.json` and `devsquad/policy.json` inside its temporary repository, with two fictional model families and fake executables. Runtime configuration uses exact locally verified models and effort settings. These examples intentionally avoid embedding today's provider model names or pretending that fixture profiles are proven. + +The CLI and MCP service must validate equivalent task objects against the same generated schema. Validate paths/refs against the temporary repository only in integration tests; ordinary schema tests validate shape without touching the filesystem. diff --git a/docs/plans/engineering-team/examples/branch-review.json b/docs/plans/engineering-team/examples/branch-review.json new file mode 100644 index 0000000..8197d11 --- /dev/null +++ b/docs/plans/engineering-team/examples/branch-review.json @@ -0,0 +1,45 @@ +{ + "schema_version": 1, + "project": { + "repo_path": "/absolute/path/to/fixture-repository", + "base_ref": "fixture-base", + "target_ref": "fixture-candidate" + }, + "workflow": "branch-review", + "goal": "Review the fixture change and report supported correctness findings.", + "task_class": "fixture-review-small", + "acceptance": [ + { + "id": "review-exact-diff", + "description": "Findings identify the resolved candidate and supporting file locations.", + "evidence_kind": "review" + }, + { + "id": "report-check-outcome", + "description": "Include the fixture check result even when it reports a failure.", + "evidence_kind": "check" + } + ], + "checks": [ + { + "id": "fixture-tests", + "argv": ["python3", "-m", "unittest", "discover", "-s", "tests"], + "cwd": ".", + "timeout_seconds": 30, + "required_to_pass": false + } + ], + "scope": {"read_paths": ["src", "tests"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json" + }, + "budget": { + "wall_seconds": 600, + "max_provider_calls": 3, + "max_revisions": 0, + "max_fallbacks_per_step": 1 + }, + "origin": {"surface": "cli"} +} diff --git a/docs/plans/engineering-team/examples/issue-delivery.json b/docs/plans/engineering-team/examples/issue-delivery.json new file mode 100644 index 0000000..02d3834 --- /dev/null +++ b/docs/plans/engineering-team/examples/issue-delivery.json @@ -0,0 +1,45 @@ +{ + "schema_version": 1, + "project": { + "repo_path": "/absolute/path/to/fixture-repository", + "base_ref": "fixture-base", + "target_ref": "fixture-base" + }, + "workflow": "issue-delivery", + "goal": "Fix the fixture parser's handling of an empty input and add a regression test.", + "task_class": "fixture-bugfix-small", + "acceptance": [ + { + "id": "empty-input", + "description": "Empty input returns the documented empty result without an exception.", + "evidence_kind": "check" + }, + { + "id": "independent-review", + "description": "A different verified model reviews the exact candidate and resolves critical findings.", + "evidence_kind": "review" + } + ], + "checks": [ + { + "id": "fixture-tests", + "argv": ["python3", "-m", "unittest", "discover", "-s", "tests"], + "cwd": ".", + "timeout_seconds": 30, + "required_to_pass": true + } + ], + "scope": {"read_paths": ["src", "tests"], "write_paths": ["src/parser.py", "tests/test_parser.py"]}, + "lead": {"mode": "headless"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json" + }, + "budget": { + "wall_seconds": 1200, + "max_provider_calls": 9, + "max_revisions": 2, + "max_fallbacks_per_step": 1 + }, + "origin": {"surface": "cli"} +} From 621789ca1e5e0c669b39a6faebcd265011fecb36 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 6 Sep 2026 12:38:18 +0530 Subject: [PATCH 003/197] docs: clarify automatic selection and adopt selective council design --- ...02-surface-independent-engineering-team.md | 4 + docs/plans/engineering-team/CONTRACTS.md | 14 +- docs/plans/engineering-team/IMPLEMENTATION.md | 12 +- .../engineering-team/SELECTION-AND-COUNCIL.md | 120 ++++++++++++++++++ docs/plans/engineering-team/START-HERE.md | 2 + docs/plans/engineering-team/backlog.json | 14 ++ .../examples/branch-review.json | 2 +- .../examples/issue-delivery.json | 2 +- 8 files changed, 164 insertions(+), 6 deletions(-) create mode 100644 docs/plans/engineering-team/SELECTION-AND-COUNCIL.md diff --git a/docs/adr/ADR-002-surface-independent-engineering-team.md b/docs/adr/ADR-002-surface-independent-engineering-team.md index 2216801..027f3ee 100644 --- a/docs/adr/ADR-002-surface-independent-engineering-team.md +++ b/docs/adr/ADR-002-surface-independent-engineering-team.md @@ -115,6 +115,8 @@ An execution profile is `(harness, harness version, model family, exact model, e Routing starts with a small versioned preference list per role/task class. Filter for capability, permission, quality eligibility and verified model/effort support. Then consider every applicable quota window, cooldown, concurrency, deadline and latency. If no eligible profile is available, block with a reason. Never silently relax required capabilities or switch to paid API usage. +Selection is automatic by default. A user may pin a validated profile for any role; unpinned roles remain automatic. The selected profile defines the permitted toolbox, and the worker selects actual tool calls within it. Overrides have explicit fallback behavior. Discovery and evaluation can propose new profiles; changing the default policy remains a reviewed, versioned decision. See the [selection and Council amendment](../plans/engineering-team/SELECTION-AND-COUNCIL.md). + Account pools span applications and repositories where the underlying allowance is shared. Provider observations have sources and expiry times; unknown allowance is unknown. DevSquad's concurrency reservations do not reserve quota with a provider. Spend estimates, token counts, characters and subscription allowance are distinct measurements. ## Learning and documentation are part of completion @@ -148,6 +150,8 @@ Runtime records generate a receipt at every terminal state and a handoff packet The detailed [contracts](../plans/engineering-team/CONTRACTS.md) and [work packages](../plans/engineering-team/IMPLEMENTATION.md) are normative for the build. Their gates replace claims based only on dry runs or syntax checks. +**Optional C1 after M6:** Adapt LLM Council's independent-proposal, critique and synthesis pattern for difficult decisions. Use a bounded council within this runner, with evidence-based judgement and saved dissent. It does not delay M7 or replace routine implementation/review/checks. The [source study and extension gate](../plans/engineering-team/SELECTION-AND-COUNCIL.md) explain the protocol and capacity tradeoff; automatic council triggering requires evaluation evidence. + ## Source notes Product integration documentation checked on 2026-09-06; recheck at installation because paths and host versions change. diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index d373c24..b4f9021 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -45,12 +45,14 @@ The Task schema rejects unknown fields and checks finite bounds. It contains: | `checks[]` | `id`, `argv` string array, repository-relative `cwd`, `timeout_seconds`, `required_to_pass` | | `scope` | Repository-relative `read_paths`, `write_paths`; non-empty writes only for delivery | | `lead` | `mode: host` or `headless`; headless requires candidate profiles in policy | -| `routing` | `profiles_file`, `policy_file`; absolute or repository-relative trusted config files | -| `budget` | `wall_seconds` of active run execution, `max_provider_calls`, `max_revisions`, `max_fallbacks_per_step`; all finite non-negative integers, wall/calls positive | +| `routing` | `profiles_file`, `policy_file`; absolute or repository-relative trusted config files; optional per-role `overrides` | +| `budget` | `wall_seconds` of active run execution, `max_worker_invocations`, `max_revisions`, `max_fallbacks_per_step`; all finite non-negative integers, wall/invocations positive | | `origin` | `surface` label, optional `session_ref`; no authorization or remote invocation implied | Snapshot the task, resolved refs, config files and their hashes before enqueue. Do not resolve a moving branch again halfway through a run. V1 uses committed inputs only: if dirty files intersect declared scope, reject with an actionable `INPUT_INVALID` rather than silently omitting them. Explicit dirty-worktree snapshot support is deferred. Validate paths against traversal and symlink escape. Task checks and policies are execution authority: repo content and provider output cannot add commands or widen permissions. +`max_worker_invocations` counts launched adapter attempts, including retries, fallback attempts and headless lead invocations. It does not count native internal model requests: one CLI worker may run several model/tool turns. Observed native requests/tokens/quota and host work have separate nullable measurements. Pre-launch validation or capacity rejection does not consume a launch, though its elapsed work still counts toward execution time. The September selection/Council amendment renames the earlier design field `max_provider_calls` before schema implementation to make this unit explicit. Enforce native turn/tool limits only where the adapter verifies support; a worker-launch cap alone is not a token, quota or spending cap. + The CLI must print the resolved scope/check plan in validation output; an app lead should supply it from the user's actual task. No magic inference from a README's embedded instructions. Checks may legitimately have side effects; run them in the isolated workspace with the declared process policy. “Read-only reviewer” does not mean executing arbitrary repository scripts is safe to treat as read-only. `Policy` contains an `id`, `version`, workflow role candidate lists, per-task-class minimum quality status, `require_different_model_for_review`, optional `prefer_different_harness_for_review`, account-pool policy and experiment budget. V1 role names are `implementer`, `reviewer`, `lead`; checks are deterministic process steps, not model calls. Add a `researcher` role only when a workflow needs specific research artifacts. Don't make every task pay for research or a council. @@ -67,6 +69,14 @@ quality_status; evidence_refs[] Install-time discovery reports supported values and evidence (`documented`, `probed`, `unavailable`, `unknown`) with `checked_at`, CLI version and toolset hash. Selecting a known catalog entry verifies that it exists, not that the invocation used it: attempts retain separate `requested` and `observed` fields. Manually verified mappings may establish identity for a versioned harness; silent model fallback must never be labelled confirmed. +### Automatic selection and manual overrides + +Default selection is automatic among policy-eligible profiles. The host supplies task requirements; the deterministic router chooses the model/effort/tool profile without another planning-model call. Within the selected toolbox, the worker chooses individual tool calls. Discovery can enumerate supported configurations; initial quality preferences and account-pool mappings still require evidence and operator setup. + +`routing.overrides` defaults to `{}`. Each key is a model role supported by the chosen workflow, with value `{profile_id, fallback}`; `fallback` defaults to `none` and optionally allows `policy`. A pin constrains the initial profile; `none` also prohibits substitution/escalation to another profile. A pinned profile must pass the ordinary identity, capability, quality, permission and billing filters. Invalid pins fail validation; temporarily unavailable pins block. `policy` permits the normal qualified fallback list within the task budget. Roles without pins remain automatic. Record overrides and their origin in the routing receipt and separate them from automatic decisions in evaluations. + +The [selection and Council amendment](SELECTION-AND-COUNCIL.md) gives examples and ownership boundaries. Evidence gathering/proposals are automatic; promotion of changed routing defaults remains a reviewed policy update. Its optional C1 workflow is outside the initial two-workflow schema until the gated extension ships. + ## 3. Adapter execution boundary Retain `invoke_codex`, `invoke_gemini`, `invoke_grok` and their legacy stdout/exit/error contract. Add a Claude headless adapter. New execution uses a bridge around shared wrapper configuration and classification helpers: diff --git a/docs/plans/engineering-team/IMPLEMENTATION.md b/docs/plans/engineering-team/IMPLEMENTATION.md index 1cafc6f..b3cb453 100644 --- a/docs/plans/engineering-team/IMPLEMENTATION.md +++ b/docs/plans/engineering-team/IMPLEMENTATION.md @@ -58,7 +58,7 @@ Work in this order: **New files:** `src/devsquad/{workflows,router,reports}.py`, `policies/`, role templates, branch-review schema/fixtures; `test/core/test_review_workflow.py`. -1. Implement profile/policy loading, version/hash snapshots and capability/permission/quality filters. Use a small static candidate list. Add basic shared-pool concurrency and typed unknown capacity now; richer observations arrive in M6. +1. Implement profile/policy loading, version/hash snapshots and capability/permission/quality filters. Select automatically from a small static candidate list, with validated per-role profile overrides and explicit fallback semantics from the selection amendment. Add basic shared-pool concurrency and typed unknown capacity now; richer observations arrive in M6. 2. Resolve commit refs and task scope; create a frozen review workspace. Reject intersecting dirty inputs rather than silently omitting them. 3. Implement reviewer → trusted checks → lead disposition, with `required_to_pass` semantics. Support host handoffs through CLI and a headless lead profile. Expose clear `awaiting_host` packets. 4. Produce receipt JSON/Markdown, events export and artifact manifest for every terminal outcome. Record unknown host usage honestly. @@ -67,6 +67,10 @@ Work in this order: **First product stop:** At this point DevSquad is usable from terminal for a bounded job. Demonstrate it before adding more architecture. +**Selection gate:** Identical task/policy/availability snapshots produce the same automatic selection. Pin one role and verify others remain automatic. Unsupported pins fail; an unavailable pin with `fallback:none` blocks without substitution. `fallback:policy` selects only a qualified alternative. No override broadens permissions or billing authority. Receipt explains effective model, effort and toolbox; a worker can only call permitted tools. Fake fixtures exercise an explicit bounded escalation and preserve each attempt's original profile. + +**Accounting gate:** `max_worker_invocations` limits adapter launches, including retries and the headless lead. Native internal model/tool turns and observed token/quota usage are separate fields; unknown stays null. A simulated worker with several native turns must not be reported as one measured model request or a fixed allowance charge. + ## M4 — Use that same run from local apps **Outcome:** Apps are interchangeable clients of the saved run. @@ -107,7 +111,7 @@ Work in this order: 1. Implement pool mapping, applicable quota windows, TTL/source/confidence and local reservations. Use documented provider observations where available and timestamped manual values otherwise. Status displays unknown and stale values explicitly. 2. Add bounded fallback, native retry metadata and explicit paid-API eligibility. Snapshot every routing decision with exclusions and policy version. -3. Add final/late outcome records, role contribution, lead repair and evidence references; produce comparison reports with sample sizes and missingness. +3. Add final/late outcome records, role contribution, lead repair and evidence references; produce comparison reports with sample sizes and missingness. Separate manually pinned decisions, normal automatic routing and experimental assignments to avoid treating selection bias as a profile improvement. 4. Implement experiment specs and held-out evaluation fixtures. `learn propose` generates a draft hypothesis/evaluation/decision packet; promotion remains a reviewed Git policy change with a rollback target. 5. Generate receipts/handoffs on run transitions, and curated documentation on explicit report/proposal operations. Record drift and affected evidence; do not introduce an unrequested scheduled automation. @@ -129,6 +133,10 @@ Work in this order: **Acceptance gate:** Fresh standalone install works without Claude installed; existing Claude plugin install contains `plugin/core` contents correctly. Reinstall creates no duplicate hook/server registration and a release update does not break a running job. Record actual start/observe/handoff-or-cancel receipts from every listed surface. Complete one end-to-end delivery with a different-model reviewer after installation. Documentation commands run as written. If an installed host cannot support an operation, retain that item as blocked with exact evidence instead of declaring universal support. +## Optional C1 — Selective Council decisions + +After M6, implement the separately gated [Council extension](SELECTION-AND-COUNCIL.md). Reuse the runner and native adapters for two independent proposals, a distinct critic and the existing lead. Preserve dissent, enforce evidence checks and account for every call. C1 does not block M7; its own acceptance/evaluation gate controls whether automatic council triggering is enabled. Track it under `extensions` in the backlog, preserving the seven core milestones. + ## Evidence and ongoing status Update [backlog.json](backlog.json) as work proceeds. Allowed states: `pending`, `in_progress`, `blocked`, `complete`. A blocked item needs a precise reason and remaining independent work; completion requires non-empty evidence. Each evidence item records `kind`, `revision`, `command_or_action`, `outcome`, `artifact`, `recorded_at` and `availability`. Use an implementation revision already committed when writing a subsequent completion receipt; do not insert a circular “this commit's hash” placeholder. diff --git a/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md new file mode 100644 index 0000000..68dd0de --- /dev/null +++ b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md @@ -0,0 +1,120 @@ +# Automatic selection and selective councils + +**Design amendment, 2026-09-06. Nothing here is implemented yet.** Clarifies the [contracts](CONTRACTS.md) and adds a separately gated Council extension to [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md). It follows the user's question about automatic selection and LLM Council. + +## Who chooses what? + +**Everyday selection is automatic. Manual selection is an override. Changes to the selection policy are reviewed.** + +| Decision | Owner | +|---|---| +| Discover installed harnesses, models, supported efforts and tools | DevSquad's discovery/probes | +| Connect accounts, identify shared allowance pools, set spending/access limits | User setup, assisted by discovery | +| Frame the task, scope, acceptance criteria and required capabilities | Current host lead, or supplied terminal task | +| Select model, effort and permitted toolbox for each role | Deterministic router applying versioned policy and current availability | +| Decide which permitted tool to call during work | Selected worker, inside the assigned permissions | +| Pin a particular configuration for this task | User override, resolved by the host into a validated profile | +| Collect outcomes and propose better configurations | DevSquad's evidence and evaluation loop | +| Promote a changed default policy | Reviewed versioned change, initially human-governed | + +You should usually say “fix this issue” or “review this branch,” not fill in a model matrix. The host prepares a task; the router selects an eligible profile. The router itself needs no model call. A profile packages an exact harness/model, native effort setting, tool access, permission policy and account pool. Selection among tested combinations keeps the search space manageable. + +```mermaid +flowchart TD + U[Task and requirements] --> O{Explicit role override?} + O -->|yes| P[Validate pinned profile] + O -->|no| F[Filter eligible profiles] + P --> Q[Check capabilities, permissions and capacity] + F --> Q + Q --> R[Use valid pin or choose from preferences] + R --> W[Run with an explanation of the choice] + W --> E{Acceptance met?} + E -->|no| X[Bounded retry / configured escalation] + X --> Q + E -->|yes| L[Record outcome and actual settings] + L --> H[Evaluate policy improvements] +``` + +Effort is selected with the profile. Easy work can use a proven lower-effort configuration; difficult work can use a stronger validated configuration. Escalation follows declared failure/rework rules and a finite budget. A higher effort label is not treated as a universal quality score, and no model gets an automatic maximum setting simply because it supports one. + +The router selects the allowed toolbox; the worker chooses actual calls within it. Required X/search/video access must exist on that harness/profile. It cannot grant an app-native capability to an API model by naming the vendor. Selecting a profile never installs tools, adds permissions or purchases API usage implicitly. + +### Override contract + +Add optional `routing.overrides`, keyed by model role. Each value contains `profile_id` and `fallback` (`none` by default, or explicit `policy`). Empty/missing overrides means automatic selection for every role. Pinning one role leaves the other roles automatic. Pinning every model role gives manual assignment. + +Illustrative fragment, using a fictional configured profile: + +```json +{ + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + "overrides": { + "reviewer": {"profile_id": "my-verified-review-profile", "fallback": "none"} + } + } +} +``` + +Validate role names against the selected workflow. A pin must meet the same capability, identity, quality, permission and billing constraints as automatic candidates. Invalid settings return a validation error; temporary unavailability blocks with a reason. `fallback:none` never silently substitutes another profile. `fallback:policy` permits only the ordinary qualified candidate list and logs the substitution. Profile edits produce a new version; no live model/effort/tool mutation mid-attempt. All surfaces submit this same task shape. + +Reports explain selections, excluded alternatives, explicit overrides, escalations and observed settings. An unmeasured or manually pinned trial must not be counted as proof of a general routing improvement. + +## What LLM Council actually contributes + +Studied **Karpathy's original repository**, pinned at [`92e1fcc`](https://github.com/karpathy/llm-council/tree/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131). This is a source review, not a performance benchmark or a survey of forks. Its author describes it as an exploratory, unsupported project. [Original README](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/README.md). + +```mermaid +flowchart LR + Q[One question] --> A[N independent answers] + A --> B[N anonymous-label peer rankings] + B --> C[Configured chairman synthesizes] +``` + +The code gives every ranker all successful answers, including its own, in the same order. The chairman sees named answers and raw critiques. Ranking uses permissive text parsing and average positions. For four members, a normal completed run makes **nine chat-completion requests**: four answers, four rankings, one synthesis. Ranker input repeats all answers, so aggregate answer-content volume grows roughly quadratically with membership. These are code-derived counts, not measured spending or quality improvements. [Council implementation](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/council.py). + +Membership/chairman are configured explicitly; the API request supplies model and messages without effort, tool definitions or account-quota selection. It uses OpenRouter API access, so running this original app would not automatically draw from your CLI subscriptions or supply their native tools. [Configuration](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/config.py), [API client](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/openrouter.py). + +Conversation files store stage outputs; the reviewed implementation has no evaluation-to-policy learning loop. The next question is sent without the stored conversation history. The first message adds a title-generation call, making the four-member initial exchange ten calls. [Storage](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/storage.py), [Request orchestration](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/backend/main.py). + +## Adopt the deliberation pattern inside DevSquad + +| Adopt | DevSquad adaptation | +|---|---| +| Independent first opinions | Same frozen brief and evidence packet; participants cannot read each other's initial proposals | +| Peer critique | Criterion-by-criterion findings, source/test references and unresolved objections | +| Anonymous presentation | Hide profile metadata; store per-judge order and label mapping; counterbalance order in evaluations | +| One synthesis role | Existing host/headless lead explains the chosen approach and retains meaningful dissent | +| Inspectable stages | Persist proposals, critiques, decisions, failures and later outcomes as ordinary run artifacts | + +My recommendation is **selective Council mode** for architectural tradeoffs, competing debugging hypotheses and consequential disputed reviews. Routine delivery keeps implementer → independent reviewer → checks. Agreement is useful evidence to examine; it cannot override failing tests or prove a claim true. Research on LLM judges documents position, verbosity and self-enhancement biases, supporting calibration rather than treating peer ranks as ground truth. [LLM-as-a-judge study](https://arxiv.org/abs/2306.05685). + +```mermaid +flowchart TD + T[Task] --> G{Configured council trigger?} + G -->|no| N[Normal engineering workflow] + G -->|yes| A[Two independent proposals] + A --> B["One independent critic
Evidence and objections"] + B --> C[Existing lead selects / synthesizes] + C --> D[Decision artifact with dissent and validation plan] + D --> N +``` + +Proposed starting budget: two proposers, one critic, and one headless lead = **four worker invocations before DevSquad-level retries**. With a host lead, there are three worker invocations plus separately recorded host work. A native CLI worker can perform several model/tool turns; these units are not comparable to the original Council's nine API requests. This smaller deliberation protocol needs its own quality/capacity evaluation. Count launches using `max_worker_invocations`, recording native usage and context bytes separately. Council participants stay read-only; it never creates several concurrent implementation writers. + +### C1 — Optional Council extension after M6 + +**Dependencies:** M3–M6 through M6. C1 is a separate pending extension and does not block M7 or first usability. Reuse the runner, adapters, artifacts, routing, cancellation and evidence store. No additional service or UI platform. + +Add a feature-gated `council-decision` workflow after the two core workflows are proven. M1–M7 schemas keep the original workflow enum until C1 is implemented; expose supported workflows through doctor. C1 adds the new enum and its strict configuration schemas together, with a documented contract revision. + +- **Inputs:** Existing task/criteria/scope/budget plus a CouncilSpec: proposer candidate profiles, critic candidate profiles, evidence artifact IDs/hashes, rubric, minimum valid proposals (two), critic requirement (one), maximum council invocations and a reason for invocation. Proposer/critic profiles use the same pin/fallback semantics as ordinary roles. The normal lead remains the sole decision authority. Stage barriers and role-scoped filesystem/MCP artifact access withhold peer proposals until every initial proposal is finalized; independence cannot rely on prompt wording alone. +- **Selection:** Automatically choose two distinct verified model identities and a critic whose model identity differs from both authors. Prefer independent families where qualified, without inferring independence from harness names. If the minimum eligible set or budget is unavailable, report the council as blocked; do not manufacture a quorum. The ordinary engineering workflow remains separately available. +- **Critique:** No self-scoring. Hide identity metadata, retain raw provenance separately, and record reproducible randomized presentation order. Anonymity is partial because writing/content can reveal identity. Require structured criterion assessments and evidence references; validate missing/duplicate/unknown candidate IDs. Avoid forcing an overall numeric rank. +- **Decision:** Save supported claims, chosen proposal or synthesis, discarded alternatives, unresolved objections and required validation. The lead cannot turn majority preference into test acceptance. Failed providers, missing evidence and critic failure are explicit; there is no silent “consensus” when participants drop out. +- **Tools:** A common baseline evidence packet makes proposals comparable. Additional native research is allowed only by profile and declared scope, with its sources retained. Unequal evidence access is recorded as a confounder for model comparisons. Check code claims through the normal trusted verification path. +- **Triggers:** Start with explicit user/lead request and a capped policy allowance. Automatic triggering remains disabled until C1's comparison gate passes. Later policy may trigger on declared architectural decisions or unresolved evidence-backed review conflicts; model self-confidence alone is not a trigger. One council per decision by default; additional rounds consume explicit allowance. +- **Learning:** Compare normal workflow vs selective council on matched/held-out cases: accepted quality, escaped defects, lead rework, latency and observed allowance. Keep question, profile, effort, prompt and evidence versions. Council rank or agreement never directly promotes a model globally. + +**Acceptance:** Proposers cannot see each other's draft artifacts; critic cannot author-score; shuffled labels map back correctly. Fake fixtures cover missing proposer, missing critic, invalid judgement IDs, empty output, quota exhaustion, cancellation/resume and recorded dissent. A seeded wrong majority cannot override a failed mandatory check. A live bounded council yields an inspectable decision using existing subscription adapters. A predeclared held-out comparison records benefit, harm or inconclusive results; automatic use stays disabled unless a reviewed policy change is supported. Receipt includes every worker attempt, failed attempts and available native usage, not just the selected answer. diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index eecabaa..d482002 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -22,6 +22,8 @@ flowchart LR 3. Work through [IMPLEMENTATION](IMPLEMENTATION.md), one milestone at a time. [backlog.json](backlog.json) is the completion record; all milestones initially have `status: pending` and empty evidence. 4. Consult the [assessment](../../audits/2026-09-06-engineering-team-assessment.md) for verified defects and history, and [ADR-001](../../adr/ADR-001-contract-and-ledger-core.md) for legacy constraints retained by ADR-002. +Selection is automatic by default, with validated per-role profile overrides. Read the [selection and LLM Council amendment](SELECTION-AND-COUNCIL.md) for the clarified contract. Its optional C1 extension follows M6 and does not block the seven core milestones. + ## Copyable execution brief ```text diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 297f638..fb000c6 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -6,6 +6,7 @@ "architecture": "../../adr/ADR-002-surface-independent-engineering-team.md", "contracts": "CONTRACTS.md", "implementation": "IMPLEMENTATION.md", + "selection_and_council": "SELECTION-AND-COUNCIL.md", "status": "planned", "next_milestone": "M1", "milestones": [ @@ -72,5 +73,18 @@ "evidence": [], "blocker": null } + ], + "extensions": [ + { + "id": "C1", + "title": "Selective evidence-based Council decisions", + "optional": true, + "depends_on": ["M6"], + "status": "pending", + "specification": "SELECTION-AND-COUNCIL.md", + "acceptance_section": "C1 — Optional Council extension after M6", + "evidence": [], + "blocker": null + } ] } diff --git a/docs/plans/engineering-team/examples/branch-review.json b/docs/plans/engineering-team/examples/branch-review.json index 8197d11..09cd3f4 100644 --- a/docs/plans/engineering-team/examples/branch-review.json +++ b/docs/plans/engineering-team/examples/branch-review.json @@ -37,7 +37,7 @@ }, "budget": { "wall_seconds": 600, - "max_provider_calls": 3, + "max_worker_invocations": 3, "max_revisions": 0, "max_fallbacks_per_step": 1 }, diff --git a/docs/plans/engineering-team/examples/issue-delivery.json b/docs/plans/engineering-team/examples/issue-delivery.json index 02d3834..3e14491 100644 --- a/docs/plans/engineering-team/examples/issue-delivery.json +++ b/docs/plans/engineering-team/examples/issue-delivery.json @@ -37,7 +37,7 @@ }, "budget": { "wall_seconds": 1200, - "max_provider_calls": 9, + "max_worker_invocations": 9, "max_revisions": 2, "max_fallbacks_per_step": 1 }, From bfaa390ceb42ecb22278d0d0c88ea150213b5091 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 6 Sep 2026 13:20:35 +0530 Subject: [PATCH 004/197] docs: add native Codex integration and model release lifecycle --- ...02-surface-independent-engineering-team.md | 6 +- docs/plans/engineering-team/CONTRACTS.md | 15 ++- docs/plans/engineering-team/IMPLEMENTATION.md | 20 +-- .../MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md | 116 ++++++++++++++++++ .../engineering-team/SELECTION-AND-COUNCIL.md | 4 +- docs/plans/engineering-team/START-HERE.md | 2 + docs/plans/engineering-team/backlog.json | 1 + 7 files changed, 148 insertions(+), 16 deletions(-) create mode 100644 docs/plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md diff --git a/docs/adr/ADR-002-surface-independent-engineering-team.md b/docs/adr/ADR-002-surface-independent-engineering-team.md index 027f3ee..ff447e1 100644 --- a/docs/adr/ADR-002-surface-independent-engineering-team.md +++ b/docs/adr/ADR-002-surface-independent-engineering-team.md @@ -36,7 +36,7 @@ Optimize accepted outcomes first, then reduce rework, latency and scarce allowan | July design | September implementation decision | |---|---| -| Bash-only coordinator | Python 3.11+ standard-library coordinator; preserve Bash 3.2 adapter compatibility | +| Bash-only coordinator | Python 3.11+ standard-library coordinator; preserve Bash 3.2 adapter compatibility and use verified native protocols where available | | JSONL as primary ledger | Transactional SQLite event ledger and projections; JSONL is an export | | Claude session always synthesizes | One explicitly selected lead: current host or a headless worker | | Provider/role aliases and implicit self fallback | Exact execution profiles; qualified fallback or an explicit blocked state | @@ -107,7 +107,7 @@ flowchart LR Start with the smaller `branch-review` template: snapshot → review → checks → disposition → receipt. Add `issue-delivery` after this is useful. Repository changes occur in an isolated worktree with one writer. Reviewers have verified read-only permissions and a frozen revision. Checks use trusted argv arrays. Patch changes invalidate prior review/test acceptance. V1 returns work for integration; it does not merge, push, deploy or publish automatically. -Preserve the existing shell wrapper API and four error prefixes. Extract shared argument-building/classification helpers for a new bridge. In new runs Python owns timeouts, cancellation, draining and reaping; the legacy wrapper keeps its separately repaired bounded invocation path. Do not nest two competing watchdogs or rewrite provider behavior in a second place. +Preserve the existing shell wrapper API and four error prefixes. Extract shared argument-building/classification helpers for CLI adapters; prefer a verified native app-server adapter for new Codex jobs. Python owns the protocol child or CLI subprocess, timeouts, cancellation, draining and reaping; the legacy wrapper keeps its separately repaired bounded invocation path. Do not nest competing watchdogs or two job coordinators. The [native adapter amendment](../plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) records the source-backed extension. ## Select configurations, measure outcomes @@ -115,7 +115,7 @@ An execution profile is `(harness, harness version, model family, exact model, e Routing starts with a small versioned preference list per role/task class. Filter for capability, permission, quality eligibility and verified model/effort support. Then consider every applicable quota window, cooldown, concurrency, deadline and latency. If no eligible profile is available, block with a reason. Never silently relax required capabilities or switch to paid API usage. -Selection is automatic by default. A user may pin a validated profile for any role; unpinned roles remain automatic. The selected profile defines the permitted toolbox, and the worker selects actual tool calls within it. Overrides have explicit fallback behavior. Discovery and evaluation can propose new profiles; changing the default policy remains a reviewed, versioned decision. See the [selection and Council amendment](../plans/engineering-team/SELECTION-AND-COUNCIL.md). +Selection is automatic by default. A user may pin a validated profile for any role; unpinned roles remain automatic. The selected profile defines the permitted toolbox, and the worker selects actual tool calls within it. Overrides have explicit fallback behavior. Stable role aliases resolve to qualified concrete profiles and are frozen per run. Discovery/evaluation can propose replacements; a reviewed update policy may authorize guarded automatic binding promotions, while policy changes remain reviewed. See the [selection amendment](../plans/engineering-team/SELECTION-AND-COUNCIL.md) and [model lifecycle](../plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md). Account pools span applications and repositories where the underlying allowance is shared. Provider observations have sources and expiry times; unknown allowance is unknown. DevSquad's concurrency reservations do not reserve quota with a provider. Spend estimates, token counts, characters and subscription allowance are distinct measurements. diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index b4f9021..639fd90 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -43,6 +43,7 @@ The Task schema rejects unknown fields and checks finite bounds. It contains: | `goal`, `task_class` | Bounded objective and comparison class, e.g. `bugfix-python-small` | | `acceptance[]` | Stable criterion `id`, concrete `description`, `evidence_kind` (`review`, `check`, `artifact`, `host`) | | `checks[]` | `id`, `argv` string array, repository-relative `cwd`, `timeout_seconds`, `required_to_pass` | +| `review` | Optional `{mode, focus}`; mode defaults to `standard`, or `adversarial`; focus is allowed only for adversarial review | | `scope` | Repository-relative `read_paths`, `write_paths`; non-empty writes only for delivery | | `lead` | `mode: host` or `headless`; headless requires candidate profiles in policy | | `routing` | `profiles_file`, `policy_file`; absolute or repository-relative trusted config files; optional per-role `overrides` | @@ -57,6 +58,8 @@ The CLI must print the resolved scope/check plan in validation output; an app le `Policy` contains an `id`, `version`, workflow role candidate lists, per-task-class minimum quality status, `require_different_model_for_review`, optional `prefer_different_harness_for_review`, account-pool policy and experiment budget. V1 role names are `implementer`, `reviewer`, `lead`; checks are deterministic process steps, not model calls. Add a `researcher` role only when a workflow needs specific research artifacts. Don't make every task pay for research or a council. +Role candidate entries are tagged references `{kind: profile|alias, id}`. A profile reference resolves to an immutable concrete configuration; an alias such as `review.deep` resolves through the versioned qualified-binding registry. Preflight snapshots the resolved profile and fallback set, plus template/binding revisions. An alias may gain a qualified replacement for new runs without changing workflow files. Explicit `routing.overrides.profile_id` remains a concrete pin and never floats. Templates and binding changes follow the [model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md). + `Profile` contains: ```text @@ -75,11 +78,11 @@ Default selection is automatic among policy-eligible profiles. The host supplies `routing.overrides` defaults to `{}`. Each key is a model role supported by the chosen workflow, with value `{profile_id, fallback}`; `fallback` defaults to `none` and optionally allows `policy`. A pin constrains the initial profile; `none` also prohibits substitution/escalation to another profile. A pinned profile must pass the ordinary identity, capability, quality, permission and billing filters. Invalid pins fail validation; temporarily unavailable pins block. `policy` permits the normal qualified fallback list within the task budget. Roles without pins remain automatic. Record overrides and their origin in the routing receipt and separate them from automatic decisions in evaluations. -The [selection and Council amendment](SELECTION-AND-COUNCIL.md) gives examples and ownership boundaries. Evidence gathering/proposals are automatic; promotion of changed routing defaults remains a reviewed policy update. Its optional C1 workflow is outside the initial two-workflow schema until the gated extension ships. +The [selection and Council amendment](SELECTION-AND-COUNCIL.md) gives examples and ownership boundaries. Evidence gathering/proposals are automatic. Initially promotions are reviewed; the [model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) permits tested model-binding promotion under an explicitly enabled, previously reviewed `guarded_auto` policy. Changing permissions, billing authority or the promotion policy itself remains reviewed. Optional C1 is outside the initial two-workflow schema until that gated extension ships. ## 3. Adapter execution boundary -Retain `invoke_codex`, `invoke_gemini`, `invoke_grok` and their legacy stdout/exit/error contract. Add a Claude headless adapter. New execution uses a bridge around shared wrapper configuration and classification helpers: +Retain `invoke_codex`, `invoke_gemini`, `invoke_grok` and their legacy stdout/exit/error contract. Add a Claude headless adapter. Adapters declare `transport: cli_exec | native_protocol`, while exposing the same normalized lifecycle/results. The CLI transport uses a bridge around shared wrapper configuration and classification helpers: 1. `prepare(request.json)` returns `LaunchSpec`: fixed manifest-selected executable, argv array, working directory, optional stdin artifact, allowlisted environment overrides, requested model/effort, parser version, permission evidence. The bridge may source bundled wrapper functions; it must not source a request-selected file. 2. Python launches that argv directly in a new process session/group, with no `shell=True` or `eval`, owns lifecycle and captures bounded stdout/stderr to files. @@ -87,6 +90,8 @@ Retain `invoke_codex`, `invoke_gemini`, `invoke_grok` and their legacy stdout/ex The new bridge does not call `_adapter_invoke`'s watchdog or write legacy JSON usage arrays. Python writes new-run telemetry once. Existing sourced callers retain their old bookkeeping. Per-run model/effort/permission overrides must not edit shared configuration. Native CLI timeouts may act as an earlier provider limit, but Python remains the sole supervisor and cleanup owner. +For Codex, prefer a version-verified `native_protocol` adapter backed by a supervisor-owned stdio app-server. The adapter prepares typed thread/review/turn requests, maps native events into core events and retains native IDs. Requests come from validated task fields and adapter code, not arbitrary caller-supplied protocol methods. Use native interruption before process cleanup; require terminal evidence. An acknowledged start/interrupt is not task completion. Native errors normalize into the same legacy-compatible taxonomy plus structured core details. A broken transport cannot trigger a second writer without reconciliation. See the [native adapter amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) for capability/version gates and retained Bash compatibility. + The four adapter error codes stay `RATE_LIMITED | AUTH_ERROR | TIMEOUT | CLI_ERROR`, with auth checked before rate. Additional **core** errors include `INPUT_INVALID`, `PROFILE_UNSUPPORTED`, `CAPABILITY_UNAVAILABLE`, `CONFLICT`, `RECOVERY_REQUIRED`, `BUDGET_EXHAUSTED`, `POLICY_DENIED`, `INTERNAL_ERROR`. Do not expand the legacy prefix enum to represent orchestration states. Execution completion is separate from deliverable validity and acceptance. Empty output, malformed required JSON, tool/permission denial or an authentication banner can invalidate a nominal exit-0 invocation. Save raw native output and the parser verdict. Prompt compliance alone does not prove a reviewer was read-only: require a verified native restriction or OS process policy; otherwise that role is unavailable. Do not use the wrappers' current blanket approval flags in new worker profiles. @@ -97,7 +102,7 @@ Workers get `DEVSQUAD_WORKER=1`, run/attempt IDs and a delegation-depth guard. D ## 4. Durable state, concurrency and recovery -SQLite schema has `projects`, `runs`, `steps`, `attempts`, `events`, `artifacts`, `claims`, `pool_observations`, `pool_reservations`, `outcomes` and `schema_migrations`. Entity IDs are opaque UUIDs; event cursors, versions, fencing tokens and migration versions are integers. UTC timestamps accompany durations measured with a monotonic clock. Mutations increment a run `version`; the append-only event and affected projections commit together in one short transaction. File artifacts are atomically finalized and hashed before a transaction references them; crash-created unreferenced files are recoverable garbage, not valid results. +SQLite schema has `projects`, `runs`, `steps`, `attempts`, `events`, `artifacts`, `claims`, `pool_observations`, `pool_reservations`, `outcomes`, `profile_templates`, `profile_bindings`, `qualification_runs` and `schema_migrations`. Entity IDs are opaque UUIDs; event cursors, versions, fencing tokens and migration versions are integers. UTC timestamps accompany durations measured with a monotonic clock. Mutations increment a run `version`; the append-only event and affected projections commit together in one short transaction. Binding updates use their own compare-and-swap version and event. File artifacts are atomically finalized and hashed before a transaction references them; crash-created unreferenced files are recoverable garbage, not valid results. ```mermaid stateDiagram-v2 @@ -175,7 +180,7 @@ devsquad/ learning/policy-changelog.md links to evidence and policy diffs ``` -`report` derives summaries from the database; `learn propose` writes draft distilled records for review. It does not change routing. Raw task content is not automatically committed or uploaded. Track redacted evidence with reproducible case IDs/hashes and an explicit `evidence_availability` value (`local`, `tracked_fixture`, `unavailable`); a local path alone is not portable proof. Policy promotion requires a reviewed, versioned change with evaluation links. During M6 use an ordinary Git diff/review for this, not another bespoke approval UI. +`report` derives summaries from the database; `learn propose` writes draft distilled records for review. It does not change routing. Raw task content is not automatically committed or uploaded. Track redacted evidence with reproducible case IDs/hashes and an explicit `evidence_availability` value (`local`, `tracked_fixture`, `unavailable`); a local path alone is not portable proof. Policy changes require review with evaluation links; use an ordinary Git diff/review during M6. A separately enabled `guarded_auto` qualification gate may update a model binding within that policy, writing an evidence/rollback receipt for each promotion. Runtime binding updates do not edit the user's Git checkout. Use this loop: @@ -184,7 +189,7 @@ Use this loop: 3. Propose one change: model, effort, prompt/context strategy, tool access or workflow. Compare like tasks and keep the remaining settings controlled or explicitly record confounders. 4. Run a small predeclared evaluation with held-out cases, budget and stopping rule. Avoid routing only hard tasks to one model and then treating unadjusted averages as model quality. 5. Promote only with supported quality evidence, acceptable rework/latency/capacity tradeoff and a rollback target; otherwise preserve the policy and record no-change. -6. Revalidate affected profiles after model/harness/prompt/tool/policy drift or escaped defects. Do not discard unrelated evidence or automatically promote a newly released model. +6. Revalidate affected profiles after model/harness/prompt/tool/policy drift or escaped defects. Do not discard unrelated evidence or promote a model on release/discovery alone. Any guarded automatic binding promotion must satisfy the versioned qualification/evaluation rules. Default experiment budget is disabled until explicitly configured, then at most 10% of eligible runs with a hard call/time cap. Most work uses the current proven policy. V1 uses human-governed static preferences, not exhaustive permutations, an automatic bandit or foundation-model fine-tuning. Report sample sizes and missingness; tiny samples justify hypotheses, not provider rankings. diff --git a/docs/plans/engineering-team/IMPLEMENTATION.md b/docs/plans/engineering-team/IMPLEMENTATION.md index b3cb453..bcd3fee 100644 --- a/docs/plans/engineering-team/IMPLEMENTATION.md +++ b/docs/plans/engineering-team/IMPLEMENTATION.md @@ -23,20 +23,22 @@ Implement sequentially through M3. M4 and M5 may proceed in parallel after agree **Existing files:** `plugin/lib/adapter.sh`, `codex-wrapper.sh`, `gemini-wrapper.sh`, `grok-wrapper.sh`, `model-catalog.sh`; `test/test_wrapper_contract.sh`, `test/test_models.sh`. -**New files:** `plugin/core/pyproject.toml`, `bin/squad`, `src/devsquad/{cli,contracts,adapters}.py`, `schemas/`, `adapters/{codex,antigravity,grok}/`; `test/core/test_contracts.py`, `test/core/test_adapters.py` and fake executables. +**New files:** `plugin/core/pyproject.toml`, `bin/squad`, `src/devsquad/{cli,contracts,adapters,codex_protocol}.py`, `schemas/`, `adapters/{codex,antigravity,grok}/`; `test/core/test_contracts.py`, `test/core/test_adapters.py` and fake executables/protocol servers. Work in this order: 1. Establish package/import layout, Python 3.11 floor, `squad --version`, initial `doctor`, strict v1 schemas and fixture loading. Record package/source fingerprints. -2. Extract reusable argv-building and classification helpers without changing sourced-wrapper signatures or error prefixes. Implement `prepare`/`classify` bridge; retain legacy telemetry only on legacy invocations. +2. Extract reusable argv-building and classification helpers without changing sourced-wrapper signatures or error prefixes. Implement `prepare`/`classify` for CLI transport; add the version-verified Codex stdio app-server transport and event normalization. Preserve legacy telemetry only on legacy invocations. Use upstream native-protocol patterns from the model lifecycle amendment without importing a second job coordinator. 3. Fix legacy portable-watchdog completion delay and descendant cleanup with timing/process assertions. Preserve Bash 3.2 and jq-absent legacy tests. Do not make legacy operation require installing Python. -4. Add per-call explicit model/effort/tool/permission settings; verify installed CLI mappings with help/documentation and bounded probes. Remove cross-family numeric tier ranking from new selection; correct legacy resolution with compatibility tests. Catalog drift produces unknown/revalidation rather than silent identity substitution. +4. Add per-call explicit model/effort/tool/permission settings; verify installed CLI mappings with help/documentation and bounded probes. Discover Codex models/efforts through app-server metadata; define stable alias/templates separately from concrete profiles. Remove cross-family numeric tier ranking from new selection; correct legacy resolution with compatibility tests. Paginated catalog refresh retains last-good data on error/incomplete responses. Catalog drift produces unknown/revalidation rather than silent identity substitution. 5. Replace the Gemini extension whitelist and whitespace splitting with scoped, bounded context enumeration. Document when native file access replaces prompt concatenation. **Acceptance gate:** Existing suite passes. Fake immediate CLI returns promptly with a 2-second timeout (target under 1 second on normal local CI); a hanging CLI and descendant are gone by timeout plus grace. Fixtures cover empty exit-0, exit-0 auth banner, denied tool, malformed output, spaces/TSX inputs, explicit unsupported effort, unavailable capability and cross-family catalog entries. Overrides do not mutate global/project config. Doctor distinguishes supported, unverified and unavailable settings. One short read-only real adapter probe validates the chosen starting profile; save a redacted receipt and version facts. **Boundary:** No learned routing, MCP, worktree edits or universal model catalog. Manifests for unprobed harnesses remain visibly unverified. +**Native/discovery gate:** A fake app-server exercises thread/review/turn events, explicit supported/unsupported effort and native IDs. Acknowledged start/interrupt is not reported as terminal completion. Paginated metadata assembles one complete snapshot; parse/auth/timeout failures preserve the previous snapshot. An added model becomes an unqualified candidate, and an unknown family cannot inherit capabilities from its name. Native model-list schema availability alone does not count as live entitlement or quality evidence. + ## M2 — Persist jobs and own their processes **Outcome:** Start/status/cancel/resume operate on the same saved job, including after the initiating shell exits. @@ -58,9 +60,9 @@ Work in this order: **New files:** `src/devsquad/{workflows,router,reports}.py`, `policies/`, role templates, branch-review schema/fixtures; `test/core/test_review_workflow.py`. -1. Implement profile/policy loading, version/hash snapshots and capability/permission/quality filters. Select automatically from a small static candidate list, with validated per-role profile overrides and explicit fallback semantics from the selection amendment. Add basic shared-pool concurrency and typed unknown capacity now; richer observations arrive in M6. +1. Implement profile/policy loading, alias binding resolution, version/hash snapshots and capability/permission/quality filters. Select automatically from a small static candidate list, with validated per-role profile overrides and explicit fallback semantics from the selection amendment. Freeze concrete profiles and fallback sets during preflight; never resolve a changed alias halfway through a run. Add basic shared-pool concurrency and typed unknown capacity now; richer observations arrive in M6. 2. Resolve commit refs and task scope; create a frozen review workspace. Reject intersecting dirty inputs rather than silently omitting them. -3. Implement reviewer → trusted checks → lead disposition, with `required_to_pass` semantics. Support host handoffs through CLI and a headless lead profile. Expose clear `awaiting_host` packets. +3. Implement reviewer → trusted checks → lead disposition, with `required_to_pass` semantics. Support host handoffs through CLI and a headless lead profile. Expose clear `awaiting_host` packets. Distinguish ordinary native review from a steerable adversarial-review prompt, recording the mode and supported controls; both remain review-only against the frozen candidate. 4. Produce receipt JSON/Markdown, events export and artifact manifest for every terminal outcome. Record unknown host usage honestly. **Acceptance gate:** An actual branch review returns actionable findings or a supported clean verdict, against recorded base/target hashes. A report-only failed check is included in a successful review; a required failed check prevents acceptance. A denied reviewer write or missing output cannot count as a valid review. A moving branch does not change the frozen run input. A second terminal claims a saved host handoff; a late completion from the first is fenced out. Verify original checkout/index/HEAD are unchanged. @@ -109,16 +111,20 @@ Work in this order: **New files:** `src/devsquad/{capacity,learning}.py`, observation/experiment/outcome schemas, learning templates; `test/core/test_capacity.py`, `test/core/test_learning.py`. -1. Implement pool mapping, applicable quota windows, TTL/source/confidence and local reservations. Use documented provider observations where available and timestamped manual values otherwise. Status displays unknown and stale values explicitly. +1. Implement pool mapping, applicable quota windows, TTL/source/confidence and local reservations. Use documented provider observations, including Codex app-server rate limits when supported, and timestamped manual values otherwise. Status displays unknown and stale values explicitly. 2. Add bounded fallback, native retry metadata and explicit paid-API eligibility. Snapshot every routing decision with exclusions and policy version. 3. Add final/late outcome records, role contribution, lead repair and evidence references; produce comparison reports with sample sizes and missingness. Separate manually pinned decisions, normal automatic routing and experimental assignments to avoid treating selection bias as a profile improvement. 4. Implement experiment specs and held-out evaluation fixtures. `learn propose` generates a draft hypothesis/evaluation/decision packet; promotion remains a reviewed Git policy change with a rollback target. 5. Generate receipts/handoffs on run transitions, and curated documentation on explicit report/proposal operations. Record drift and affected evidence; do not introduce an unrequested scheduled automation. +6. Implement the [model lifecycle](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md): budgeted qualification, limited trials, reviewed or explicitly enabled guarded automatic binding promotion, compare-and-swap binding versions, last-qualified fallback and rollback. Promotions stay within allowed templates, quality criteria, permissions and billing authority. They write local decision receipts and affect new runs only. The default remains reviewed until evaluation gates and policy enable guarded automation. + **Acceptance gate:** Two projects sharing one pool obey a fresh exhausted weekly window despite available short-window capacity. Stale/unknown values never become zero; external usage changes do not get assigned to one worker. Concurrency reservations release only after ownership is reconciled. A paid API fallback is excluded unless allowed. A failed original attempt later repaired by another model produces final success without crediting the original as independently successful. A late escaped bug updates outcome history. A one-variable fixture experiment produces a traceable no-change or promotion proposal, with all failures and a rollback version; insufficient evidence leaves active policy unchanged. Re-run a held-out fixture after policy change and exercise rollback. **Boundary:** No automatic learned router. Experiment budgets default off; activate only through explicit versioned policy. +**Release gate:** A new model cannot become default from discovery alone. Insufficient evidence, unsupported effort or widened permissions/billing prevents automatic promotion. Changed metadata revalidates only affected profiles; same-ID backing changes retain unknown revision when unobservable. An approved binding update affects a newly started run while an existing run and concrete override remain pinned. Concurrent promotions conflict on stale binding versions. A regression reverts to an available qualified binding and records why. All qualification/trial launches share the configured experiment and account-pool budgets. + ## M7 — Package, migrate and prove every requested surface **Outcome:** A fresh install has one known runtime, reliable documentation and real smoke evidence. @@ -128,7 +134,7 @@ Work in this order: 1. Package the core inside `plugin/`, with a stable standalone launcher and optional MCP environment. Preserve active run release pins. Installer is idempotent and reports source/plugin/standalone drift. 2. Keep legacy plugin mode available during migration. Reconcile hook registration only when duplicate evidence exists; September review found no current duplicate. New hooks call the shared route source after its tests pass and remain fast/network-free. No Python dependency imposed on legacy mode. 3. Create generated command/schema examples and concise install/operate/recover guides. Mark implemented vs deferred features; link completion claims to receipts. Keep historical ADR/audit statements dated. -4. Add offline CI for legacy and core suites (Bash 3.2/macOS compatibility and chosen Python floor/current version), package-content checks, and optional MCP tests. Live provider/app tests remain explicit bounded smoke runs. +4. Add offline CI for legacy and core suites (Bash 3.2/macOS compatibility and chosen Python floor/current version), package-content checks, native protocol compatibility fixtures and optional MCP tests. Live provider/app tests remain explicit bounded smoke runs. Document supported protocol ranges, capability drift and the optional native Claude→Codex session-import path, while retaining portable artifact handoffs for every host. 5. Verify terminal CLI, Codex App, Claude Code App local Code tab, Antigravity local IDE/CLI and Grok Build against the same saved runtime. Check native capabilities/profile identity as used, not by brand inference. **Acceptance gate:** Fresh standalone install works without Claude installed; existing Claude plugin install contains `plugin/core` contents correctly. Reinstall creates no duplicate hook/server registration and a release update does not break a running job. Record actual start/observe/handoff-or-cancel receipts from every listed surface. Complete one end-to-end delivery with a different-model reviewer after installation. Documentation commands run as written. If an installed host cannot support an operation, retain that item as blocked with exact evidence instead of declaring universal support. diff --git a/docs/plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md b/docs/plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md new file mode 100644 index 0000000..de72939 --- /dev/null +++ b/docs/plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md @@ -0,0 +1,116 @@ +# Model releases without routine DevSquad code updates + +**Design amendment, 2026-09-06; implementation pending.** Builds on [automatic selection](SELECTION-AND-COUNCIL.md), [contracts](CONTRACTS.md) and [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md). Incorporate into M1/M3/M6/M7; this adds no prerequisite platform or new core milestone. + +## What to adopt from OpenAI's Claude Code plugin + +Reviewed [openai/codex-plugin-cc](https://github.com/openai/codex-plugin-cc/tree/db52e28f4d9ded852ab3942cea316258ae4ef346), pinned at `db52e28f4d9ded852ab3942cea316258ae4ef346`. Source inspection only: no installation, account changes, session imports or live model requests. + +| Source pattern | DevSquad application | +|---|---| +| App-server thread/turn operations and structured events | Prefer Codex's native protocol for new jobs; preserve the legacy shell wrapper | +| Native review and a separate steerable adversarial review | Distinguish ordinary defect review from a targeted challenge; record review mode | +| Stored job status/result and native thread IDs | Keep DevSquad run IDs plus native thread/turn IDs for recovery and reopening | +| Native turn interruption | Interrupt the owned turn before bounded process cleanup; confirm termination | +| Model/effort overrides with native defaults when omitted | Discover defaults as metadata; resolve a verified run profile before execution | + +The implementation uses `review/start` for ordinary review and `turn/start` with effort/output schema for task execution. It also supports native thread resume and session import. These are concrete capabilities to reuse through an adapter. [Native integration source](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/plugins/codex/scripts/lib/codex.mjs). + +The plugin distinguishes read-only reviews from delegated editing and provides structured findings plus instructions to preserve verdict/evidence boundaries. DevSquad should retain this distinction while applying the user's already-authorized workflow: a delivery task can include bounded corrections, whereas review-only scope cannot. [Review schema](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/plugins/codex/schemas/review-output.schema.json), [Result handling](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/plugins/codex/skills/codex-result-handling/SKILL.md). + +The plugin's optional stop-time review gate can repeatedly return work to Claude; its README warns of extended loops and usage drain. DevSquad's existing task-scoped review gate, finite revisions and budgets remain the design. The plugin also uses the user's local Codex authentication/configuration and contributes to the same Codex limits. It does not create another allowance pool. [README](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/README.md). + +Native Claude→Codex transcript import is an optional convenience when supported. It does not imply every host can import every other host's history. Keep portable artifact handoffs as the common interface. The source resolves the selected transcript under the real Claude projects directory before importing. [Transfer path validation](https://github.com/openai/codex-plugin-cc/blob/db52e28f4d9ded852ab3942cea316258ae4ef346/plugins/codex/scripts/lib/claude-session-transfer.mjs). + +### Adapter amendment + +The new adapter abstraction supports `transport: cli_exec | native_protocol`. Both expose discovery, validated preparation, execution events, interruption, result normalization and optional resume. Codex should use `native_protocol` when the installed app-server passes conformance. Other providers keep `cli_exec` unless they expose a verified suitable native interface. + +For Codex, use a **DevSquad-owned stdio app-server child** inside the existing per-run supervisor. Avoid adding the upstream plugin's separate shared broker/job database as a second coordinator. Python owns that child's lifecycle; the adapter sends typed protocol requests. Store native thread/turn IDs and wait for actual completion notifications; accepting a start or interrupt request is not completion. Handle server requests with the declared role policy, never blanket-approve them. Use the isolated run worktree as native cwd and validate effective permissions. Protocol failure must not silently launch a duplicate CLI writer. + +Generate/test protocol schemas against supported installed versions. Keep experimental methods behind capability checks. An explicit fallback transport must satisfy the same role/model/effort contract and be selected before launch, or after safe reconciliation. Standard `review/start` and custom `turn/start` do not necessarily accept identical effort controls; unsupported explicit combinations fail visibly. The Python coordinator remains unchanged in purpose; Bash callers keep their existing four-prefix API. + +If implementation reuses upstream code, retain Apache-2.0 license/NOTICE obligations and pin provenance. Adopting behavior does not require vendoring the entire plugin. The protocol patterns are the immediate benefit; DevSquad still owns cross-provider scheduling, outcome evidence and worktree concurrency. + +## Stable names above changing model releases + +Workflows refer to stable **profile aliases**, for example `implement.balanced`, `review.deep`, `research.current`. These are DevSquad policy names, not guesses about provider tiers. Each alias binds to a tested concrete execution profile. + +```mermaid +flowchart TD + W[Stable role alias
review.deep] --> B[Current qualified binding] + B --> P[Exact harness + model + effort + tools] + C[Live provider catalog] --> N[New / changed candidates] + N --> V[Compatibility checks + bounded evaluation] + V --> G{Promotion rules satisfied?} + G -->|yes| B + G -->|no / insufficient evidence| K[Keep known working binding] +``` + +| Layer | Changes when | +|---|---| +| Workflow and role requirements | Your engineering process changes | +| Adapter code/protocol mapping | A CLI, protocol or permission interface changes | +| Model catalog and effort/tool metadata | Models or account-visible capabilities change | +| Alias → concrete profile binding | A replacement qualifies under the update policy | + +**Routine model releases should be data updates.** A changed CLI/protocol or unsupported capability may still require an adapter update. This reduces maintenance; it cannot make third-party interfaces permanently stable. + +### Discovery, not name guessing + +Codex app-server documents `model/list` with supported/default reasoning efforts, modality and upgrade metadata, and `account/rateLimits/read` for quota observations. Its local generated schema also exposes these method types. This replaces DevSquad's current hardcoded `codex: unlistable` assumption. The upstream Claude plugin's execution code and the app-server's catalog API are distinct evidence sources; do not claim the plugin itself implements a model-release router. [Official app-server documentation](https://learn.chatgpt.com/docs/app-server). + +Local verification used `codex app-server generate-json-schema` with the installed CLI and inspected `ModelListResponse`, `ThreadStartParams`, `TurnStartParams` and `GetAccountRateLimitsResponse`. This verifies available schema shapes, not account entitlement, model quality or a successful live model invocation. + +DevSquad already has a [catalog implementation](../../../plugin/lib/model-catalog.sh) and detached refresh. Extend that intent, replacing numeric/keyword tier ranking with structured identity and compatibility. A model version number is not comparable across families, and a provider's recommended default/upgrade is a candidate recommendation, not a benchmark result. + +Discovery contract: + +- Cache per harness, installed version, authenticated account identity reference and configuration scope. Use native structured metadata first; otherwise a version-tested CLI parser or curated manifest. API catalogs cannot establish subscription-CLI entitlement. +- Refresh outside hooks on first use, stale cache (initial TTL 24 hours), a changed CLI/config fingerprint, or a relevant model-not-found signal. Deduplicate refreshes with a lease, timeout, backoff and pagination. No permanent polling service is required. +- Keep the last successful snapshot on timeout, auth failure, parse error or an incomplete response. Record staleness and errors separately. A failed refresh is never a mass-removal event. Confirm removal with a complete scoped catalog or an explicit model-unavailable response, distinguishing account/auth problems. +- Store exact IDs, provider family when verified, native effort options/defaults, modalities, tool evidence, deprecation/upgrade hints, source and timestamps. Missing fields remain unknown. Do not infer tools or reasoning levels from a model name. +- Generate a small number of candidate profiles from **existing allowed templates**. Start with the template's supported effort intent and at most one experimental alternative. Do not enumerate every model × effort × tool permutation. Unknown families or widened capabilities require a policy/template change. +- A provider alias can change behind the same public ID. Record reported effective identity/revision when available; otherwise label the backing revision unknown and use metadata drift plus periodic behavior checks. Reproducibility is best-effort when providers expose no immutable revision. + +## Review the update rules once; automate routine promotion + +This refines the earlier “update defaults after review” statement: **review can authorize a bounded promotion policy once**, rather than require a human to approve every qualifying model release. Start with reviewed promotions while calibrating the evaluation; enable guarded automation after those checks prove useful. + +```mermaid +flowchart LR + D[Discover automatically] --> T[Compatibility smoke checks] + T --> E[Bounded held-out evaluation] + E --> C[Limited eligible trial] + C --> P{Preauthorized gate} + P -->|pass| A[Promote binding for new runs] + P -->|fail / unknown| R[Retain or revert binding] + A --> O{Regression / incompatible drift?} + O -->|yes| R + O -->|no| K[Keep qualified binding] +``` + +`model_updates.mode` is `reviewed` initially, or opt-in `guarded_auto`. A versioned update policy specifies allowed alias/templates/harnesses/families, evaluation cases/rubric, minimum evidence, quality non-inferiority criteria, critical-defect rule, latency/usage tolerances, candidate/trial budgets, rollback target and stop conditions. These are task-class-specific thresholds, not a global model leaderboard. Missing thresholds or insufficient evidence mean no automatic promotion. + +Metadata refresh and candidate proposals are automatic. Inference-based probes/evaluations/trials require an enabled, bounded qualification budget, share the existing experiment allowance and respect account-pool limits. `guarded_auto` cannot silently enable paid APIs, new tools/permissions, a new account route or a different policy. Those changes remain reviewed. A policy may permit a low-risk candidate trial; high-risk work keeps the qualified incumbent until the gate passes. Ordinary model-launch counts and native usage remain distinct. + +Promotion atomically updates an alias binding under a policy revision, records the previous binding and evidence IDs, and affects **new runs only**. Every run freezes the resolved concrete profile/fallback set before its worker starts. Active runs and pinned-profile overrides never move just because a new catalog appears. Retired/unavailable incumbents use an already-qualified fallback or block; do not substitute an untested “latest” model. Rollback also affects new attempts only after ownership is reconciled, and chooses an available qualified predecessor rather than a removed model. + +Only revalidate affected combinations after effort/tool/protocol drift. Evidence remains attached to the original fingerprint; passing a previous model's evaluation is not inherited automatically. Reduce trial scope and retain the incumbent when budgets or usable cases are limited. Automatic promotion is a static gate over measured evidence, not an unconstrained learned router or a self-reported confidence score. + +### State and documentation + +Keep catalog snapshots and immutable concrete profiles in the runtime registry. Add `profile_templates`, `profile_bindings` and `qualification_runs` to the transactional store. Project policy references stable aliases and allowed templates; the local binding ledger determines the current qualified concrete profile for that installation/account. Resolve and snapshot binding versions with task preflight; idempotent start replay never re-resolves them. + +Each binding change writes a local decision receipt: alias, old/new concrete IDs and fingerprints, effective/unknown settings, source release/catalog, evaluation/trial evidence, measured limits/usage, policy gate, rollback target and actor (`human` or `guarded_auto`). Derived Markdown reports summarize discovered, trialled, adopted, rejected and retired profiles. Curated exports remain available for Git documentation; runtime promotions do not edit or auto-commit the user's checkout. No new schedule or automation is installed by this design. + +## Additions to existing milestone gates + +| Milestone | Addition | +|---|---| +| M1 | Native/CLI transport contract; Codex app-server metadata; last-good catalog, templates and alias schema; unsupported effort handling | +| M3 | Resolve stable aliases to immutable profiles, explain selection and snapshot bindings; standard/adversarial review distinction | +| M6 | Qualification budget, reviewed/guarded-auto promotion, evidence receipts and rollback; real Codex quota observations when available | +| M7 | Compatibility fixtures across supported CLI/protocol versions and installation drift; document optional native session handoff | + +Required cases: paginated discovery; incomplete/error catalog retains last-good entries; added model does not become default; changed effort support revalidates only affected profiles; same-ID drift retains uncertainty; alias promotion affects new runs only; exact pin never floats; insufficient trial evidence blocks promotion; permissions/billing changes cannot auto-promote; failed/retired incumbent uses only qualified fallback; concurrent promotions use compare-and-swap binding versions; regression rolls back and produces a decision receipt. Native adapter tests cover server disconnect, interruption acknowledgment without termination, isolated cwd/permissions and resuming a still-active turn without a duplicate writer. diff --git a/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md index 68dd0de..52ba697 100644 --- a/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md +++ b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md @@ -15,7 +15,7 @@ | Decide which permitted tool to call during work | Selected worker, inside the assigned permissions | | Pin a particular configuration for this task | User override, resolved by the host into a validated profile | | Collect outcomes and propose better configurations | DevSquad's evidence and evaluation loop | -| Promote a changed default policy | Reviewed versioned change, initially human-governed | +| Promote a changed default policy | Reviewed versioned change; it may preauthorize bounded model-binding updates | You should usually say “fix this issue” or “review this branch,” not fill in a model matrix. The host prepares a task; the router selects an eligible profile. The router itself needs no model call. A profile packages an exact harness/model, native effort setting, tool access, permission policy and account pool. Selection among tested combinations keeps the search space manageable. @@ -61,6 +61,8 @@ Validate role names against the selected workflow. A pin must meet the same capa Reports explain selections, excluded alternatives, explicit overrides, escalations and observed settings. An unmeasured or manually pinned trial must not be counted as proof of a general routing improvement. +The [model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) adds stable aliases, automatic discovery and bounded qualification. After calibration, an enabled `guarded_auto` policy can promote tested bindings without per-release manual edits. The reviewed policy remains the authority; discovery or council votes alone cannot promote a candidate. + ## What LLM Council actually contributes Studied **Karpathy's original repository**, pinned at [`92e1fcc`](https://github.com/karpathy/llm-council/tree/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131). This is a source review, not a performance benchmark or a survey of forks. Its author describes it as an exploratory, unsupported project. [Original README](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/README.md). diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index d482002..a419670 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -24,6 +24,8 @@ flowchart LR Selection is automatic by default, with validated per-role profile overrides. Read the [selection and LLM Council amendment](SELECTION-AND-COUNCIL.md) for the clarified contract. Its optional C1 extension follows M6 and does not block the seven core milestones. +Also read the [native adapters and model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md): use verified Codex app-server capabilities, stable profile aliases, automatic catalog updates and qualified binding promotions. These refine M1/M3/M6/M7; they add no prerequisite milestone and do not require rewriting workflows for each model release. + ## Copyable execution brief ```text diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index fb000c6..614a4e9 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -7,6 +7,7 @@ "contracts": "CONTRACTS.md", "implementation": "IMPLEMENTATION.md", "selection_and_council": "SELECTION-AND-COUNCIL.md", + "model_lifecycle_and_native_adapters": "MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md", "status": "planned", "next_milestone": "M1", "milestones": [ From ff1fa602d546c9b19c50ae32fc00d7066a7aaf10 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 6 Sep 2026 17:15:59 +0530 Subject: [PATCH 005/197] docs: add full Sol implementation and usability handoff --- docs/plans/engineering-team/SOL-HANDOFF.md | 104 +++++++++++++++++++++ docs/plans/engineering-team/START-HERE.md | 6 +- docs/plans/engineering-team/backlog.json | 2 + 3 files changed, 110 insertions(+), 2 deletions(-) create mode 100644 docs/plans/engineering-team/SOL-HANDOFF.md diff --git a/docs/plans/engineering-team/SOL-HANDOFF.md b/docs/plans/engineering-team/SOL-HANDOFF.md new file mode 100644 index 0000000..5877aff --- /dev/null +++ b/docs/plans/engineering-team/SOL-HANDOFF.md @@ -0,0 +1,104 @@ +# Sol execution handoff — build, test and make DevSquad usable + +This is the full execution prompt for Sol. It is an implementation assignment; the underlying runtime is still pending at handoff creation. Copy this document into Sol, or ask Sol to read this file and execute it in full. + +## Objective and persistence + +Build the complete DevSquad engineering-team plan into a simple, reliable local product I can actually use from terminal, Codex App, Claude Code App, Antigravity and Grok Build. Treat this as a persistent implementation goal. Execute, test, repair and document it through completion; do not stop after planning, scaffolding, M1, the first successful demo or passing existing tests. + +Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. +Architecture branch at handoff: `codex/surface-independent-team-plan`. +Latest architecture amendment before this handoff: `bfaa390`. + +Inspect current Git state and files first. Work from a checkout containing this handoff and all plan amendments; GitHub `main` may not contain these local commits. Preserve existing work. Create or continue an implementation branch under `codex/` from this state; do not reset to an older baseline. Read repository instructions and `CONTRIBUTING.md`. + +## Read the complete specification + +Read these files relative to the repository, in order: + +1. `docs/plans/engineering-team/START-HERE.md` +2. `docs/adr/ADR-002-surface-independent-engineering-team.md` +3. `docs/plans/engineering-team/CONTRACTS.md` +4. `docs/plans/engineering-team/IMPLEMENTATION.md` +5. `docs/plans/engineering-team/SELECTION-AND-COUNCIL.md` +6. `docs/plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md` +7. `docs/plans/engineering-team/backlog.json` and `docs/plans/engineering-team/examples/` + +Use `docs/audits/2026-09-06-engineering-team-assessment.md` for verified starting defects/history. ADR-001 and the old `.planning/` records are historical context; they do not supersede the September decisions or prove implementation completion. Recheck source and installed provider capabilities instead of trusting dated model examples. + +**Scope clarification:** This assignment includes M1–M7 and C1. Council is optional to invoke in the product, but its implementation and acceptance gates are included in this delivery. M3 is the first usable checkpoint, not the finish line. Later amendments override older conflicting details. The simple task-entry requirement below extends the earlier deferral of natural-language task preparation; it does not authorize a general workflow engine or a second planner. + +Resolve ordinary implementation choices yourself. Where current evidence requires a contract change, make the smallest coherent amendment and update affected schemas, docs and tests together. Do not restart the architecture exercise or quietly remove hard requirements to obtain green tests. + +## Deliver the whole engineering team + +- Implement the shared Python runner, transactional event/state store, isolated worktrees, durable jobs and common CLI/MCP service. Preserve legacy Bash 3.2 wrapper callers, jq-absent behavior and the four error prefixes. +- Use a verified native Codex app-server adapter, keeping legacy CLI compatibility. Support Claude, Antigravity and Grok through tested adapters. Discover actual models, efforts, tools and permissions from each installed harness; preserve requested versus observed settings and unknown values. +- Deliver branch review and bounded issue implementation → different-model review → correction → tests → lead disposition. Bind acceptance evidence to the exact candidate. Keep one writer per worktree, one lead per run and bounded retries. Implement ordinary and adversarial review distinctly. +- Make every listed local surface operate on the same saved runs. Closing a client must not lose work. Provide status, results, events, cancel, safe resume and fenced host handoffs. Native transcript import is a capability-gated convenience; portable artifact handoffs are required across hosts. +- Route automatically among eligible model/effort/tool profiles, with exact per-role pins and explicit fallback. Account for shared subscription pools, every applicable quota window, latency, unknown/stale observations and local concurrency. Maximize verified outcomes and useful capacity; do not force every provider into every task. +- Implement stable profile aliases, automatic catalog refresh, last-good snapshot retention, bounded qualification/trials, guarded promotion and rollback. Model discovery alone must not change defaults. Freeze run bindings and concrete pins. Ship/test `guarded_auto` as an opt-in after calibration; permission or billing changes stay outside it. +- Record all attempts, failures, fallbacks, reviewer contributions, lead repairs and later corrections. Generate receipts, handoffs, evaluation/decision reports and ongoing documentation. Separate worker invocations from native internal model calls and quota usage. Proposed improvements need comparable evidence. +- Implement C1 using independent proposals, a distinct critic and the existing lead. Preserve dissent, evidence and budget limits. Test manual Council use; keep automatic triggering disabled until its evaluation gate and policy authorize it. + +## Make ordinary use simple + +Ship a guided local setup that discovers existing installations/authentication, explains readiness and configures DevSquad's required local integration without asking me to maintain a model matrix. Preserve unrelated app settings and credentials. Provide a clean readiness report when a capability cannot be verified. + +Deliver this small human-facing command surface over the same service and contracts: + +```text +squad setup +squad doctor +squad review --base main +squad fix "the bounded issue to resolve" +squad council "the decision to evaluate" +squad status [RUN] +squad result [RUN] +squad cancel RUN +squad resume RUN +``` + +These are target commands to implement, not commands that exist at handoff creation. Preserve the specified low-level JSON/API operations for automation. Optional omitted run IDs resolve only when the current project has an unambiguous relevant run; otherwise present the choices. Never silently target an unrelated job. + +Normal review/fix/council entry must not require manually writing JSON. In an app, the current lead constructs the task. In terminal, use the configured single headless lead, where necessary, for one bounded task-preparation step against approved templates and allowed checks. Record that preparation, its budget and output. Validate criteria, scope and commands before execution; generated task text cannot expand permission or spending authority. Reuse the same planning authority rather than spawning a second lead. Retain committed-input constraints unless explicitly amended with equivalent tested snapshot guarantees. + +Show the selected roles/profiles, why they were chosen, progress, the run ID and a clear next action. Provide one verified command per normal operation. Errors should say what failed, whether work is still running and how to recover. Keep model IDs, raw schemas and protocol details out of the normal user flow unless they help resolve an issue. + +Do not add a dashboard, cloud service, arbitrary DAG builder or extra configuration layer merely to present these features. + +## Execute in small verified increments + +Follow the dependency graph. Demonstrate M3 early, then keep going through M4/M5/M6/M7 and C1. Delegate bounded implementation/review/test tasks when useful; isolate concurrent edits and keep ownership clear. A delegated failure does not remove the requirement: continue independent work and use available execution paths. + +Use offline fake CLIs/protocol servers for development and fault injection. Use existing authenticated subscription harnesses for bounded real smoke tests and the required live acceptance demonstrations. Recheck current official capabilities at integration time. Do not purchase credits, consume usage resets, silently switch to paid APIs, publish, push, merge, deploy, send external messages or change unrelated account settings under this assignment. + +Local implementation dependencies, an isolated DevSquad installation and required local MCP registrations are part of delivery. Make them reversible/idempotent and preserve unrelated configuration. Handle authentication through normal provider flows. If quota/authentication or an unavailable host blocks a live gate, record the exact blocker, finish all independent work, and ask only for the missing action needed to complete that gate. Do not hammer a limited provider or substitute fixture results for live proof. + +## Test behavior thoroughly + +Derive a requirement-to-evidence matrix before implementing. Preserve every milestone gate in that matrix; tests must establish behavior, not mirror code structure. Run the existing `bash test/run.sh` before every commit as required, plus meaningful new tests for the changed behavior. The historical 177 assertions are a regression baseline, not evidence that the new product works. + +Required coverage includes: + +1. **Adapters:** argv/path handling including spaces and TSX; model/effort validation; auth/rate/timeout classification; empty/malformed/denied exit-0 results; native protocol events, disconnects and interruption; permissions and recursion guards. +2. **Durability:** concurrent idempotent starts, conflicting bodies, transactional events/artifacts, migrations, supervisor crash with a live child, reused PID protection, no duplicate writer, cancel/reap, restart/resume and stale host claims. Use real controlled subprocesses where fake clocks cannot prove cleanup. +3. **Engineering outcomes:** seeded defects detected by independent review, bounded repairs, mandatory failing tests blocking acceptance, report-only failures remaining visible, stale-patch evidence rejection, scope enforcement and preservation of the user's checkout. +4. **Routing/capacity:** deterministic selection from identical snapshots; exact pins/fallback; two projects sharing one pool; short and weekly windows; stale/unknown data; external account consumption; blocked paid-API fallback; native usage versus worker-launch counts. +5. **Model lifecycle/learning:** pagination, incomplete discovery preserving last-good data, affected-profile revalidation, same-ID uncertainty, qualified promotion, insufficient evidence, new-run-only binding changes, rollback, failed attempts later repaired and escaped-defect feedback. +6. **Council:** isolated first proposals, no author judging their own proposal, valid label mapping, missing participants, invalid critiques, recorded dissent and votes unable to override objective failure. Run the predeclared comparison and keep automatic use disabled when evidence is inconclusive. +7. **Packaging and real hosts:** fresh standalone install without Claude, plugin package contents, reinstall/update without duplicate hooks/MCP servers, active runs surviving package/client changes, and actual operation from terminal, Codex App, Claude Code App local Code tab, Antigravity and Grok Build. Parsing a config or passing an MCP unit test does not prove an app integration. + +Run one genuine bounded issue through at least two harnesses with a different verified review model. Save exact candidate/check/review evidence. Exercise a real cross-surface start → observe → handoff/finish and a cancel/recovery flow. Each provider adapter and each named local surface needs its own supported-operation smoke receipt; do not force all providers into one job to satisfy that coverage. + +Do a fresh-install usability walkthrough using only the quickstart and normal commands, without hand-editing task JSON. Fix setup friction, misleading status, confusing errors and documentation commands that fail. Obtain an independent implementation/UX review when available and resolve actionable findings; do not claim an independent review that did not occur. + +## Completion and handoff back to me + +Update `backlog.json` after each verified checkpoint with revision, command/action, result and portable redacted evidence. Keep a requirement matrix and concise implementation status beside it. Commit small verified increments, never stash work, and end with a clean tree. Keep secrets/raw private task content out of tracked evidence. + +Before declaring completion, audit the full requested scope against the actual installed product. Fix failures and rerun the affected checks. A partial implementation, unavailable live gate or unsupported required host is incomplete even if offline tests pass. Record precise residual blockers rather than weakening the gate or marking it done. Do not wait indefinitely when no process is live; continue remaining independent work. + +Deliver working code and local setup; reproducible tests and live receipts; an updated architecture/contract reference; a concise quickstart with one workflow visual; and a short recovery/troubleshooting guide. My final summary should state what works, exact commands to start using it, the evidence location, remaining limitations and the implementation branch/commits. Prefer a compact readiness table over a long narrative. + +Start by inspecting the current state and implementing the earliest unmet dependency. Continue until this whole assignment is complete or the remaining requirements are explicitly blocked by external state that you cannot resolve. diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index a419670..8d6f67a 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -2,6 +2,8 @@ **Build status: planned, not implemented.** This packet follows the local/GitHub review and Dikshant's September 6 brief. Start with M1; do not run another open-ended architecture exercise. +**Full-build assignment:** Use [SOL-HANDOFF.md](SOL-HANDOFF.md) for the user's request to have Sol execute everything, test thoroughly and make normal use simple. It includes M1–M7 plus the opt-in Council feature, and adds guided task entry over the same contracts. + > Build an AI engineering team that can be operated from terminal, Codex, Claude Code, Antigravity and Grok. Use each eligible model/effort/tool configuration where it produces the best verified outcome; account for shared subscription limits. Preserve work across surfaces, learn from attempts and maintain the evidence automatically. ```mermaid @@ -32,7 +34,7 @@ Also read the [native adapters and model lifecycle amendment](MODEL-LIFECYCLE-AN Implement DevSquad's September engineering-team plan in this repository. Read docs/plans/engineering-team/START-HERE.md and its contracts first. Start at the earliest pending milestone whose dependencies are complete. -For the initial pass, implement M1 and pass its acceptance gate before M2. +Implement M1 and pass its gate, then continue through M2–M7 and C1. Preserve existing Bash 3.2 wrapper callers and their four error prefixes. Keep all distributable core files inside plugin/core; add no cloud service. Use fake CLIs for development; real provider runs are bounded smoke tests. @@ -40,7 +42,7 @@ Do not change global AI account settings or silently switch to paid APIs. Before marking a milestone complete, record the revision, checks, outcome, and a reproducible receipt in backlog.json and update the relevant docs. Run bash test/run.sh before each commit as CONTRIBUTING.md requires. -Commit each verified milestone. Continue through ready work when requested; +Commit each verified milestone. Continue through all ready work; report exact blockers rather than claiming unsupported integrations work. ``` diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 614a4e9..b134d22 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -8,6 +8,8 @@ "implementation": "IMPLEMENTATION.md", "selection_and_council": "SELECTION-AND-COUNCIL.md", "model_lifecycle_and_native_adapters": "MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md", + "execution_brief": "SOL-HANDOFF.md", + "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "planned", "next_milestone": "M1", "milestones": [ From 55e93a2d2477f52d20850254c51e90b9d1550cc0 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 6 Sep 2026 17:24:02 +0530 Subject: [PATCH 006/197] docs: consolidate branches and establish the Sol build starting point --- .../audits/2026-09-06-branch-consolidation.md | 62 +++++++++++++++++++ docs/plans/engineering-team/SOL-HANDOFF.md | 5 +- docs/plans/engineering-team/START-HERE.md | 2 + 3 files changed, 67 insertions(+), 2 deletions(-) create mode 100644 docs/audits/2026-09-06-branch-consolidation.md diff --git a/docs/audits/2026-09-06-branch-consolidation.md b/docs/audits/2026-09-06-branch-consolidation.md new file mode 100644 index 0000000..be78538 --- /dev/null +++ b/docs/audits/2026-09-06-branch-consolidation.md @@ -0,0 +1,62 @@ +# Branch consolidation — September 6, 2026 + +The repository now has two working branches, with matching names locally and on GitHub. Runtime implementation is still pending; branch cleanup does not complete any implementation milestone. + +```mermaid +flowchart LR + M[main: published runtime] --> E[codex/engineering-team: architecture + Sol build] + E --> V[Verified milestone commits] + V --> R[Review completed implementation] + R --> P[Merge to main when authorized] + A[February legacy history] --> T[Retained tags + local recovery bundle] +``` + +## Canonical branches + +| Branch | Purpose | Starting evidence | +|---|---|---| +| `main` | Published runtime baseline; tracks `origin/main` | `fa68f68621dcad166ab24cf4b3db379ecbe819ac` | +| `codex/engineering-team` | Complete architecture, execution prompt and subsequent Sol implementation | Contains all five design commits through `ff1fa602d546c9b19c50ae32fc00d7066a7aaf10`, followed by this cleanup record | + +Use [SOL-HANDOFF.md](../plans/engineering-team/SOL-HANDOFF.md) as the full build assignment. Start on the build branch or a worktree based on it. Before implementation, require a clean tree and `git merge-base --is-ancestor ff1fa60 HEAD` to succeed. GitHub's default branch remains `main`, so a default-branch checkout alone does not contain the plan. + +## What was sorted + +| Previous branch | Finding | Disposition | +|---|---|---| +| `codex/surface-independent-team-plan` | Latest complete plan and handoff, five commits ahead of `main` | Renamed to `codex/engineering-team`, published with upstream tracking | +| `codex/engineering-team-assessment` | `c7f5930`, already contained in the plan branch | Redundant local branch removed; commit retained in build history | +| `holdout-protocol` | `044dd75`, fully merged; `main` is 20 commits ahead and zero behind | Redundant local and GitHub branches removed; commit retained in `main` | +| `backup-bug-fixes` | `f6d9c4c`, 85 commits in unrelated February history, no merge base with current `main` | Branch removed after verified complete backup; existing local tag `v0.1.1-bug-fixes` retains the exact tip | + +The old bug-fix lineage was reviewed by Sol before implementation. Its meaningful fixes were ported by `b70fc22`, an ancestor of current `main`, and subsequently evolved in the current plugin. Do not cherry-pick the old production snapshot: its duplicated packaging and older wrappers would regress the present structure. Historical audit documents remain accessible through the tag. + +Existing tags `v0.1.1-bug-fixes`, `v1.0` and `v1.1` were retained unchanged. Only `v1.0` was already on GitHub; the two local historical tags and recovery bundle remain local. Cleanup does not create a release or publish the separate legacy history. + +## Recovery + +A complete, verified Git bundle was created before any branch deletion: + +```text +~/.codex/backups/devsquad/2026-09-06-before-branch-cleanup-ff1fa60.bundle +``` + +The bundle preserves all original branch refs, tags and reachable history. It is outside the repository and is not a cloud backup. Restore an old branch only when actually needed: + +```bash +git branch backup-bug-fixes v0.1.1-bug-fixes +git branch holdout-protocol 044dd75 +git branch codex/engineering-team-assessment c7f5930 +``` + +If local tags are unavailable, recover the legacy branch from the bundle: + +```bash +git fetch "$HOME/.codex/backups/devsquad/2026-09-06-before-branch-cleanup-ff1fa60.bundle" refs/heads/backup-bug-fixes:refs/heads/backup-bug-fixes +``` + +## Ongoing branch discipline + +Keep milestone checkpoints on `codex/engineering-team`. Use separate worktrees only for concurrent work, then integrate their verified commits and remove their temporary branch refs/worktrees. Preserve unique work before removal. Keep `main` as the reviewed baseline and merge the completed build only when authorized. Fetch/prune before branch cleanup; check ancestry rather than interpreting a branch name or date as proof that it is obsolete. + +The cleanup audit found one working tree, no stashes, no open or historical GitHub pull requests, and no remote changes after fetching. Validation includes ancestry checks, complete bundle verification, the existing offline test suite, clean working-tree checks and exact local/GitHub branch-tip comparison. These checks establish repository hygiene; they do not establish that the planned engineering-team runtime works. diff --git a/docs/plans/engineering-team/SOL-HANDOFF.md b/docs/plans/engineering-team/SOL-HANDOFF.md index 5877aff..ee2b8a6 100644 --- a/docs/plans/engineering-team/SOL-HANDOFF.md +++ b/docs/plans/engineering-team/SOL-HANDOFF.md @@ -7,10 +7,11 @@ This is the full execution prompt for Sol. It is an implementation assignment; t Build the complete DevSquad engineering-team plan into a simple, reliable local product I can actually use from terminal, Codex App, Claude Code App, Antigravity and Grok Build. Treat this as a persistent implementation goal. Execute, test, repair and document it through completion; do not stop after planning, scaffolding, M1, the first successful demo or passing existing tests. Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. -Architecture branch at handoff: `codex/surface-independent-team-plan`. +Canonical build branch after cleanup: `codex/engineering-team` (local and GitHub). +Complete architecture and original handoff checkpoint: `ff1fa60`. Latest architecture amendment before this handoff: `bfaa390`. -Inspect current Git state and files first. Work from a checkout containing this handoff and all plan amendments; GitHub `main` may not contain these local commits. Preserve existing work. Create or continue an implementation branch under `codex/` from this state; do not reset to an older baseline. Read repository instructions and `CONTRIBUTING.md`. +Inspect current Git state and files first. Continue `codex/engineering-team`, or use an isolated Codex worktree based on that branch. Verify `git merge-base --is-ancestor ff1fa60 HEAD` and preserve existing work. GitHub `main` remains the published runtime baseline and does not contain this build plan. Do not restart from `main`, resurrect an archived branch or merge the unrelated February backup history. Read the [branch consolidation record](../../audits/2026-09-06-branch-consolidation.md), repository instructions and `CONTRIBUTING.md`. Use one integration branch for milestone checkpoints; remove temporary task branches/worktrees after their work is integrated and preserved. ## Read the complete specification diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index 8d6f67a..fe221e6 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -4,6 +4,8 @@ **Full-build assignment:** Use [SOL-HANDOFF.md](SOL-HANDOFF.md) for the user's request to have Sol execute everything, test thoroughly and make normal use simple. It includes M1–M7 plus the opt-in Council feature, and adds guided task entry over the same contracts. +**Git starting point:** `codex/engineering-team` is the shared local/GitHub build branch. `main` remains the published runtime baseline. The [branch consolidation record](../../audits/2026-09-06-branch-consolidation.md) documents the preserved history and recovery paths. Continue this branch, or base a Codex worktree on it; verify the complete handoff checkpoint `ff1fa60` is an ancestor before coding. + > Build an AI engineering team that can be operated from terminal, Codex, Claude Code, Antigravity and Grok. Use each eligible model/effort/tool configuration where it produces the best verified outcome; account for shared subscription limits. Preserve work across surfaces, learn from attempts and maintain the evidence automatically. ```mermaid From 572e452c0076934d9824ce2b42c3b57d2287a610 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 6 Sep 2026 18:31:37 +0530 Subject: [PATCH 007/197] feat: establish truthful M1 invocation core --- .gitignore | 4 + plugin/core/adapters/antigravity/adapter.json | 10 ++ plugin/core/adapters/codex/adapter.json | 12 ++ plugin/core/adapters/grok/adapter.json | 10 ++ plugin/core/bin/squad | 8 + plugin/core/build/lib/devsquad/__init__.py | 3 + plugin/core/build/lib/devsquad/adapters.py | 151 ++++++++++++++++ plugin/core/build/lib/devsquad/catalog.py | 60 +++++++ plugin/core/build/lib/devsquad/cli.py | 85 +++++++++ .../core/build/lib/devsquad/codex_protocol.py | 76 +++++++++ plugin/core/build/lib/devsquad/contracts.py | 103 +++++++++++ plugin/core/build/lib/devsquad/validation.py | 53 ++++++ plugin/core/profiles/templates.json | 9 + plugin/core/pyproject.toml | 25 +++ plugin/core/schemas/adapter.schema.json | 15 ++ .../schemas/execution-identity.schema.json | 13 ++ plugin/core/schemas/launch-spec.schema.json | 18 ++ .../schemas/normalized-result.schema.json | 19 +++ plugin/core/schemas/policy.schema.json | 6 + plugin/core/schemas/profile.schema.json | 20 +++ plugin/core/schemas/task.schema.json | 11 ++ plugin/core/src/devsquad/__init__.py | 3 + plugin/core/src/devsquad/adapters.py | 161 ++++++++++++++++++ plugin/core/src/devsquad/catalog.py | 60 +++++++ plugin/core/src/devsquad/cli.py | 87 ++++++++++ plugin/core/src/devsquad/codex_protocol.py | 76 +++++++++ plugin/core/src/devsquad/contracts.py | 103 +++++++++++ plugin/core/src/devsquad/validation.py | 66 +++++++ plugin/lib/adapter.sh | 33 +++- plugin/lib/gemini-wrapper.sh | 80 +++++++-- plugin/lib/model-catalog.sh | 39 +++-- test/core/test_m1.py | 142 +++++++++++++++ test/test_m1_legacy.sh | 84 +++++++++ test/test_models.sh | 41 ++++- 34 files changed, 1653 insertions(+), 33 deletions(-) create mode 100644 plugin/core/adapters/antigravity/adapter.json create mode 100644 plugin/core/adapters/codex/adapter.json create mode 100644 plugin/core/adapters/grok/adapter.json create mode 100755 plugin/core/bin/squad create mode 100644 plugin/core/build/lib/devsquad/__init__.py create mode 100644 plugin/core/build/lib/devsquad/adapters.py create mode 100644 plugin/core/build/lib/devsquad/catalog.py create mode 100644 plugin/core/build/lib/devsquad/cli.py create mode 100644 plugin/core/build/lib/devsquad/codex_protocol.py create mode 100644 plugin/core/build/lib/devsquad/contracts.py create mode 100644 plugin/core/build/lib/devsquad/validation.py create mode 100644 plugin/core/profiles/templates.json create mode 100644 plugin/core/pyproject.toml create mode 100644 plugin/core/schemas/adapter.schema.json create mode 100644 plugin/core/schemas/execution-identity.schema.json create mode 100644 plugin/core/schemas/launch-spec.schema.json create mode 100644 plugin/core/schemas/normalized-result.schema.json create mode 100644 plugin/core/schemas/policy.schema.json create mode 100644 plugin/core/schemas/profile.schema.json create mode 100644 plugin/core/schemas/task.schema.json create mode 100644 plugin/core/src/devsquad/__init__.py create mode 100644 plugin/core/src/devsquad/adapters.py create mode 100644 plugin/core/src/devsquad/catalog.py create mode 100644 plugin/core/src/devsquad/cli.py create mode 100644 plugin/core/src/devsquad/codex_protocol.py create mode 100644 plugin/core/src/devsquad/contracts.py create mode 100644 plugin/core/src/devsquad/validation.py create mode 100644 test/core/test_m1.py create mode 100755 test/test_m1_legacy.sh diff --git a/.gitignore b/.gitignore index cd007e9..95bb347 100644 --- a/.gitignore +++ b/.gitignore @@ -6,6 +6,10 @@ # Dependencies node_modules/ +__pycache__/ +*.py[cod] +.pytest_cache/ +*.egg-info/ # Logs *.log diff --git a/plugin/core/adapters/antigravity/adapter.json b/plugin/core/adapters/antigravity/adapter.json new file mode 100644 index 0000000..d97db19 --- /dev/null +++ b/plugin/core/adapters/antigravity/adapter.json @@ -0,0 +1,10 @@ +{ + "schema_version": 1, + "name": "antigravity", + "transport": "cli_exec", + "binary_candidates": ["agy", "antigravity"], + "model_provider": null, + "capabilities": {"efforts_by_model": {}, "native_model_list": false, "resume": true}, + "permission_profiles": {"read_only": ["--sandbox"], "workspace_write": []}, + "output_format": "json" +} diff --git a/plugin/core/adapters/codex/adapter.json b/plugin/core/adapters/codex/adapter.json new file mode 100644 index 0000000..9a1d757 --- /dev/null +++ b/plugin/core/adapters/codex/adapter.json @@ -0,0 +1,12 @@ +{ + "schema_version": 1, + "name": "codex", + "transport": "native_protocol", + "fallback_transport": "cli_exec", + "binary_candidates": ["codex"], + "model_provider": "openai", + "verified_harness_versions": ["codex-cli 0.135.0"], + "capabilities": {"efforts_by_model": {}, "native_model_list": true, "resume": true}, + "permission_profiles": {"read_only": [], "workspace_write": []}, + "output_format": "jsonl" +} diff --git a/plugin/core/adapters/grok/adapter.json b/plugin/core/adapters/grok/adapter.json new file mode 100644 index 0000000..94f8162 --- /dev/null +++ b/plugin/core/adapters/grok/adapter.json @@ -0,0 +1,10 @@ +{ + "schema_version": 1, + "name": "grok", + "transport": "cli_exec", + "binary_candidates": ["grok"], + "model_provider": "xai", + "capabilities": {"efforts_by_model": {}, "native_model_list": false, "resume": true}, + "permission_profiles": {"read_only": ["--disable-web-search"], "workspace_write": []}, + "output_format": "json" +} diff --git a/plugin/core/bin/squad b/plugin/core/bin/squad new file mode 100755 index 0000000..3d8a4dd --- /dev/null +++ b/plugin/core/bin/squad @@ -0,0 +1,8 @@ +#!/usr/bin/env python3 +from pathlib import Path +import sys + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) +from devsquad.cli import main + +raise SystemExit(main()) diff --git a/plugin/core/build/lib/devsquad/__init__.py b/plugin/core/build/lib/devsquad/__init__.py new file mode 100644 index 0000000..22a4c4a --- /dev/null +++ b/plugin/core/build/lib/devsquad/__init__.py @@ -0,0 +1,3 @@ +"""DevSquad's surface-independent local core.""" + +__version__ = "0.1.0" diff --git a/plugin/core/build/lib/devsquad/adapters.py b/plugin/core/build/lib/devsquad/adapters.py new file mode 100644 index 0000000..152c27e --- /dev/null +++ b/plugin/core/build/lib/devsquad/adapters.py @@ -0,0 +1,151 @@ +"""Manifest-driven M1 adapter preparation and output classification.""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import subprocess +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from .contracts import ContractError, ExecutionIdentity, LaunchSpec, NormalizedResult, ProfileUnsupported, SCHEMA_VERSION + +ERROR_PATTERNS = ( + ("AUTH_ERROR", re.compile(r"auth|unauthorized|ineligible|\b401\b|\b403\b", re.I)), + ("RATE_LIMITED", re.compile(r"rate.?limit|quota|resource.?exhausted|too many requests|\b429\b", re.I)), +) + + +@dataclass(frozen=True) +class AdapterManifest: + name: str + transport: str + binary_candidates: tuple[str, ...] + model_provider: str | None + efforts_by_model: dict[str, tuple[str, ...]] + permission_profiles: dict[str, tuple[str, ...]] + output_format: str + + @classmethod + def load(cls, path: Path) -> "AdapterManifest": + raw = json.loads(path.read_text()) + required = {"schema_version", "name", "transport", "binary_candidates", "capabilities"} + missing = required - raw.keys() + if missing or raw["schema_version"] != 1: + raise ContractError(f"invalid adapter manifest: missing={sorted(missing)}") + capabilities = raw["capabilities"] + return cls( + name=raw["name"], transport=raw["transport"], + binary_candidates=tuple(raw["binary_candidates"]), + model_provider=raw.get("model_provider"), + efforts_by_model={k: tuple(v) for k, v in capabilities.get("efforts_by_model", {}).items()}, + permission_profiles={k: tuple(v) for k, v in raw.get("permission_profiles", {}).items()}, + output_format=raw.get("output_format", "text"), + ) + + def resolve_binary(self) -> str | None: + return next((p for name in self.binary_candidates if (p := shutil.which(name))), None) + + +def _permission_args(manifest: AdapterManifest, permission: str) -> tuple[str, ...]: + try: + return manifest.permission_profiles[permission] + except KeyError as exc: + raise ContractError(f"unsupported permission profile: {permission}") from exc + + +def prepare_cli( + manifest: AdapterManifest, *, prompt: str, cwd: str, model: str | None, + effort: str | None, permission: str, timeout_seconds: int, stdin_path: str | None = None, +) -> LaunchSpec: + binary = manifest.resolve_binary() + if not binary: + raise ContractError(f"adapter unavailable: {manifest.name}") + if effort is not None: + supported = manifest.efforts_by_model.get(model or "") + if supported is None or effort not in supported: + raise ProfileUnsupported(f"unsupported or unverified effort {effort!r} for {manifest.name} model {model!r}") + args: list[str] + if manifest.name == "codex": + sandbox = "read-only" if permission == "read_only" else "workspace-write" + args = [binary, "exec", "--json", "--cd", cwd, "--sandbox", sandbox] + if model: + args += ["--model", model] + if effort: + args += ["-c", f'model_reasoning_effort="{effort}"'] + args += [prompt] + elif manifest.name == "antigravity": + args = [binary, "--print", prompt, "--output-format", "json", "--mode", "plan" if permission == "read_only" else "accept-edits"] + if model: + args += ["--model", model] + if effort: + args += ["--effort", effort] + elif manifest.name == "grok": + args = [binary, "--single", prompt, "--output-format", "json", "--cwd", cwd, "--permission-mode", "plan" if permission == "read_only" else "acceptEdits", "--no-subagents"] + if model: + args += ["--model", model] + if effort: + args += ["--reasoning-effort", effort] + else: + raise ContractError(f"no argv builder for adapter: {manifest.name}") + args.extend(_permission_args(manifest, permission)) + requested = ExecutionIdentity( + harness=manifest.name, harness_version=None, model_provider=manifest.model_provider, + model_family=None, model=model, effort=effort, permissions=permission, + verification="unverified", + ) + return LaunchSpec(SCHEMA_VERSION, manifest.name, "cli_exec", tuple(args), str(Path(cwd).resolve()), stdin_path, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) + + +def _provider_records(stdout: str) -> tuple[list[dict[str, Any]], bool, str | None]: + records: list[dict[str, Any]] = [] + deliverable = False + error_text = None + for line in stdout.splitlines(): + if not line.strip(): + continue + item = json.loads(line) + if not isinstance(item, dict): + raise json.JSONDecodeError("record is not an object", line, 0) + records.append(item) + if item.get("is_error") is True or item.get("error"): + error_text = str(item.get("error") or item.get("result") or item.get("message")) + kind = item.get("type") + if kind in {"result", "assistant_message", "turn.completed"} and item.get("is_error") is not True: + payload = item.get("result") or item.get("message") or item.get("text") + deliverable = isinstance(payload, str) and bool(payload.strip()) + return records, deliverable, error_text + + +def classify_cli(spec: LaunchSpec, *, returncode: int, stdout: str, stderr: str, timed_out: bool = False) -> NormalizedResult: + code = next((code for code, pattern in ERROR_PATTERNS if pattern.search(stderr)), None) + status = "succeeded" + if code == "AUTH_ERROR" or code == "RATE_LIMITED": + status = "failed" + elif timed_out or returncode in (124, 137, 143): + status, code = "timed_out", "TIMEOUT" + elif returncode != 0: + status, code = "failed", "CLI_ERROR" + elif not stdout.strip(): + status, code = "malformed", "CLI_ERROR" + elif spec.adapter in {"codex", "antigravity", "grok"}: + try: + _, deliverable, provider_error = _provider_records(stdout) + if provider_error: + code = next((candidate for candidate, pattern in ERROR_PATTERNS if pattern.search(provider_error)), "CLI_ERROR") + status = "denied" if re.search(r"permission denied|tool (?:use )?denied|not allowed", provider_error, re.I) else "failed" + elif not deliverable: + status, code = "malformed", "CLI_ERROR" + except json.JSONDecodeError: + status, code = "malformed", "CLI_ERROR" + return NormalizedResult(SCHEMA_VERSION, status, code, stdout if stdout else None, "unknown", "not_evaluated", spec.requested, None) + + +def harness_version(binary: str) -> str | None: + try: + return subprocess.run([binary, "--version"], text=True, capture_output=True, timeout=3, check=False).stdout.strip() or None + except (OSError, subprocess.TimeoutExpired): + return None diff --git a/plugin/core/build/lib/devsquad/catalog.py b/plugin/core/build/lib/devsquad/catalog.py new file mode 100644 index 0000000..59bc94c --- /dev/null +++ b/plugin/core/build/lib/devsquad/catalog.py @@ -0,0 +1,60 @@ +"""Structured, last-good model catalog handling for M1.""" + +from __future__ import annotations + +import hashlib +import json +import os +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable + +from .contracts import ContractError + + +def model_fingerprint(harness: str, version: str | None, model: dict[str, Any]) -> str: + stable = {"harness": harness, "version": version, "model": model} + return hashlib.sha256(json.dumps(stable, sort_keys=True, separators=(",", ":")).encode()).hexdigest() + + +def normalize_models(harness: str, version: str | None, models: Iterable[dict[str, Any]]) -> list[dict[str, Any]]: + normalized = [] + for raw in models: + model_id = raw.get("id") or raw.get("model") + if not isinstance(model_id, str) or not model_id: + raise ContractError("catalog model missing id") + effort_values = raw.get("supportedReasoningEfforts") or raw.get("supported_reasoning_efforts") or [] + efforts = [item.get("reasoningEffort") if isinstance(item, dict) else item for item in effort_values] + normalized.append({ + "id": model_id, + "display_name": raw.get("displayName") or raw.get("display_name") or model_id, + "family": raw.get("family"), + "default_effort": raw.get("defaultReasoningEffort") or raw.get("default_reasoning_effort"), + "supported_efforts": efforts, + "modalities": raw.get("inputModalities") or raw.get("input_modalities") or [], + "is_default": bool(raw.get("isDefault") or raw.get("is_default")), + "qualification": "unqualified", + "fingerprint": model_fingerprint(harness, version, raw), + }) + return normalized + + +def update_last_good(path: Path, *, harness: str, version: str | None, models: Iterable[dict[str, Any]] | None, complete: bool, error: str | None = None) -> dict[str, Any]: + old = json.loads(path.read_text()) if path.exists() else None + now = datetime.now(timezone.utc).isoformat() + if error or not complete or models is None: + if old: + old["last_refresh"] = {"at": now, "status": "error" if error else "incomplete", "error": error} + _atomic_json(path, old) + return old + raise ContractError(error or "catalog response incomplete and no last-good snapshot exists") + value = {"schema_version": 1, "harness": harness, "harness_version": version, "fetched_at": now, "complete": True, "models": normalize_models(harness, version, models), "last_refresh": {"at": now, "status": "ok", "error": None}} + _atomic_json(path, value) + return value + + +def _atomic_json(path: Path, value: dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + tmp = path.with_name(f"{path.name}.tmp.{os.getpid()}") + tmp.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") + os.replace(tmp, path) diff --git a/plugin/core/build/lib/devsquad/cli.py b/plugin/core/build/lib/devsquad/cli.py new file mode 100644 index 0000000..e1be5bc --- /dev/null +++ b/plugin/core/build/lib/devsquad/cli.py @@ -0,0 +1,85 @@ +"""Small M1 command surface: version, doctor, prepare and classify.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +from . import __version__ +from .adapters import AdapterManifest, classify_cli, harness_version, prepare_cli +from .contracts import ContractError, envelope, error_payload + +SOURCE_ROOT = Path(__file__).resolve().parents[2] +CORE_ROOT = SOURCE_ROOT if (SOURCE_ROOT / "adapters").is_dir() else Path(sys.prefix) / "share" / "devsquad" + + +def manifests() -> list[tuple[Path, AdapterManifest]]: + return [(path, AdapterManifest.load(path)) for path in sorted((CORE_ROOT / "adapters").glob("*/adapter.json"))] + + +def command_doctor(_: argparse.Namespace) -> tuple[dict, int]: + rows = [] + for path, manifest in manifests(): + binary = manifest.resolve_binary() + rows.append({"adapter": manifest.name, "transport": manifest.transport, "status": "unverified" if binary else "unavailable", "binary": binary, "version": harness_version(binary) if binary else None, "manifest": str(path)}) + ready = any(row["status"] != "unavailable" for row in rows) + return envelope(data={"core_version": __version__, "ready": ready, "adapters": rows}), 0 if ready else 1 + + +def command_prepare(args: argparse.Namespace) -> dict: + manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") + spec = prepare_cli(manifest, prompt=args.prompt, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) + return envelope(data=spec.to_dict()), 0 + + +def command_classify(args: argparse.Namespace) -> dict: + manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") + spec = prepare_cli(manifest, prompt="classification", cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) + result = classify_cli(spec, returncode=args.returncode, stdout=Path(args.stdout_file).read_text(), stderr=Path(args.stderr_file).read_text()) + return envelope(data=result.to_dict()), 0 + + +class ContractParser(argparse.ArgumentParser): + def error(self, message: str) -> None: + raise ContractError(message) + + +def parser() -> argparse.ArgumentParser: + p = ContractParser(prog="squad") + p.add_argument("--version", action="version", version=f"squad {__version__}") + sub = p.add_subparsers(dest="command", required=True) + doctor = sub.add_parser("doctor"); doctor.add_argument("--json", action="store_true"); doctor.set_defaults(func=command_doctor) + for name, fn in (("prepare", command_prepare), ("classify", command_classify)): + cmd = sub.add_parser(name) + cmd.add_argument("adapter", choices=("codex", "antigravity", "grok")) + cmd.add_argument("--cwd", default=str(Path.cwd())) + cmd.add_argument("--model") + cmd.add_argument("--effort") + cmd.add_argument("--permission", choices=("read_only", "workspace_write"), default="read_only") + cmd.add_argument("--timeout", type=int, default=90) + if name == "prepare": + cmd.add_argument("--prompt", required=True) + else: + cmd.add_argument("--returncode", type=int, required=True) + cmd.add_argument("--stdout-file", required=True) + cmd.add_argument("--stderr-file", required=True) + cmd.set_defaults(func=fn) + return p + + +def main(argv: list[str] | None = None) -> int: + try: + args = parser().parse_args(argv) + response, code = args.func(args) + print(json.dumps(response, sort_keys=True)) + return code + except (ContractError, OSError, json.JSONDecodeError) as exc: + code = getattr(exc, "code", "INPUT_INVALID") + print(json.dumps(envelope(error=error_payload(code, str(exc))), sort_keys=True)) + return 64 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/build/lib/devsquad/codex_protocol.py b/plugin/core/build/lib/devsquad/codex_protocol.py new file mode 100644 index 0000000..e7eedd2 --- /dev/null +++ b/plugin/core/build/lib/devsquad/codex_protocol.py @@ -0,0 +1,76 @@ +"""Typed JSON-RPC preparation/state normalization for Codex app-server. + +This module does not spawn or supervise the server. M2 gives it a connected +stdio stream owned by the run supervisor. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any + +from .contracts import ContractError + + +def request(request_id: int, method: str, params: dict[str, Any] | None = None) -> dict[str, Any]: + return {"jsonrpc": "2.0", "id": request_id, "method": method, "params": params or {}} + + +def initialize_request(request_id: int = 1) -> dict[str, Any]: + return request(request_id, "initialize", {"clientInfo": {"name": "devsquad", "version": "0.1.0"}, "capabilities": {"experimentalApi": True}}) + + +def model_list_request(request_id: int, cursor: str | None = None, limit: int = 100) -> dict[str, Any]: + params: dict[str, Any] = {"limit": limit} + if cursor: + params["cursor"] = cursor + return request(request_id, "model/list", params) + + +def turn_start_request(request_id: int, *, thread_id: str, prompt: str, model: str | None, effort: str | None, cwd: str, sandbox: str) -> dict[str, Any]: + params: dict[str, Any] = {"threadId": thread_id, "input": [{"type": "text", "text": prompt}], "cwd": cwd, "permissions": sandbox} + if model: + params["model"] = model + if effort: + params["effort"] = effort + return request(request_id, "turn/start", params) + + +@dataclass +class NativeTurnState: + thread_id: str | None = None + turn_id: str | None = None + terminal: bool = False + interrupted_acknowledged: bool = False + output: list[str] = field(default_factory=list) + events: list[dict[str, Any]] = field(default_factory=list) + + def consume(self, message: dict[str, Any]) -> None: + if not isinstance(message, dict): + raise ContractError("native message must be an object") + self.events.append(message) + method = message.get("method", "") + params = message.get("params") or message.get("result") or {} + if method in {"thread/started", "thread/start/completed"}: + self.thread_id = params.get("thread", {}).get("id") or params.get("threadId") or self.thread_id + if method in {"turn/started", "turn/start/completed"}: + self.turn_id = params.get("turn", {}).get("id") or params.get("turnId") or self.turn_id + if method in {"item/agentMessage/delta", "turn/output/delta"}: + self.output.append(params.get("delta", "")) + if method in {"turn/interrupt/completed", "turn/interrupted/acknowledged"}: + self.interrupted_acknowledged = True + if method in {"turn/completed", "review/completed", "turn/failed", "turn/interrupted"}: + self.terminal = True + + +def parse_model_page(response: dict[str, Any]) -> tuple[list[dict[str, Any]], str | None]: + if "error" in response: + raise ContractError(f"model/list failed: {response['error']}") + result = response.get("result") + if not isinstance(result, dict): + raise ContractError("model/list missing result") + models = result.get("data") or result.get("models") + if not isinstance(models, list): + raise ContractError("model/list is incomplete") + cursor = result.get("nextCursor") or result.get("next_cursor") + return models, cursor diff --git a/plugin/core/build/lib/devsquad/contracts.py b/plugin/core/build/lib/devsquad/contracts.py new file mode 100644 index 0000000..9fca7dc --- /dev/null +++ b/plugin/core/build/lib/devsquad/contracts.py @@ -0,0 +1,103 @@ +"""Strict M1 contracts for prepared invocations and normalized results.""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Any, Literal + +SCHEMA_VERSION = 1 +Transport = Literal["cli_exec", "native_protocol"] +Verification = Literal["verified", "unverified", "unavailable", "unknown"] + + +class ContractError(ValueError): + """Raised before launch when an invocation contract is unsupported.""" + + code = "INPUT_INVALID" + + +class ProfileUnsupported(ContractError): + code = "PROFILE_UNSUPPORTED" + + +@dataclass(frozen=True) +class ExecutionIdentity: + harness: str + harness_version: str | None + model_provider: str | None + model_family: str | None + model: str | None + effort: str | None + tools: tuple[str, ...] = () + permissions: str = "read_only" + account_pool: str | None = None + verification: Verification = "unknown" + + +@dataclass(frozen=True) +class LaunchSpec: + """A launch description. M2 owns spawning, timeout, cancellation and reaping.""" + + schema_version: int + adapter: str + transport: Transport + argv: tuple[str, ...] + cwd: str + stdin_path: str | None + timeout_seconds: int + requested: ExecutionIdentity + environment: dict[str, str] = field(default_factory=dict) + + def __post_init__(self) -> None: + if self.schema_version != SCHEMA_VERSION: + raise ContractError("unsupported schema_version") + if self.transport not in ("cli_exec", "native_protocol"): + raise ContractError("unsupported transport") + if not self.argv or not all(isinstance(v, str) and v for v in self.argv): + raise ContractError("argv must be a non-empty string array") + if self.timeout_seconds <= 0: + raise ContractError("timeout_seconds must be positive") + if not Path(self.cwd).is_absolute(): + raise ContractError("cwd must be absolute") + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +@dataclass(frozen=True) +class NormalizedResult: + """Execution evidence without conflating artifacts or acceptance.""" + + schema_version: int + execution_status: Literal[ + "succeeded", "failed", "timed_out", "interrupted", "denied", "malformed" + ] + error_code: str | None + output: str | None + artifact_status: Literal["present", "missing", "not_required", "unknown"] + acceptance_status: Literal["pending", "accepted", "rejected", "not_evaluated"] + requested: ExecutionIdentity + observed: ExecutionIdentity | None + native_ids: dict[str, str] = field(default_factory=dict) + events: tuple[dict[str, Any], ...] = () + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +def envelope(*, data: Any = None, error: dict[str, Any] | None = None) -> dict[str, Any]: + if (data is None) == (error is None): + raise ContractError("exactly one of data and error is required") + return {"schema_version": SCHEMA_VERSION, "ok": error is None, "data": data, "error": error} + + +def error_payload(code: str, message: str, *, retryable: bool = False, details: dict[str, Any] | None = None) -> dict[str, Any]: + return {"code": code, "message": message, "retryable": retryable, "details": details or {}} + + +def validate_launch_payload(value: dict[str, Any]) -> None: + expected = {"schema_version", "adapter", "transport", "argv", "cwd", "stdin_path", "timeout_seconds", "requested", "environment"} + if set(value) != expected: + raise ContractError(f"LaunchSpec fields differ: {sorted(set(value) ^ expected)}") + LaunchSpec(**{**value, "argv": tuple(value["argv"]), "requested": ExecutionIdentity(**{**value["requested"], "tools": tuple(value["requested"]["tools"])})}) diff --git a/plugin/core/build/lib/devsquad/validation.py b/plugin/core/build/lib/devsquad/validation.py new file mode 100644 index 0000000..dd0c585 --- /dev/null +++ b/plugin/core/build/lib/devsquad/validation.py @@ -0,0 +1,53 @@ +"""Dependency-free strict validation for M1 public fixtures.""" + +from __future__ import annotations +from pathlib import Path +from typing import Any +from .contracts import ContractError + +TASK_FIELDS = {"schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin", "review"} + + +def _exact(value: dict[str, Any], allowed: set[str], required: set[str], label: str) -> None: + if not isinstance(value, dict): + raise ContractError(f"{label} must be an object") + unknown, missing = set(value) - allowed, required - set(value) + if unknown or missing: + raise ContractError(f"{label} fields invalid: unknown={sorted(unknown)} missing={sorted(missing)}") + + +def _relative(path: str, label: str) -> None: + p = Path(path) + if p.is_absolute() or ".." in p.parts: + raise ContractError(f"{label} must be repository-relative without traversal") + + +def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) -> None: + _exact(value, TASK_FIELDS, TASK_FIELDS - {"review"}, "task") + if value["schema_version"] != 1 or value["workflow"] not in {"branch-review", "issue-delivery"}: + raise ContractError("unsupported task schema or workflow") + project = value["project"] + _exact(project, {"repo_path", "base_ref", "target_ref"}, {"repo_path", "base_ref", "target_ref"}, "project") + if not Path(project["repo_path"]).is_absolute() or (require_existing_repo and not (Path(project["repo_path"]) / ".git").exists()): + raise ContractError("project.repo_path must be an existing absolute Git repository") + if not isinstance(value["goal"], str) or not value["goal"].strip() or not isinstance(value["task_class"], str) or not value["task_class"].strip(): + raise ContractError("goal and task_class must be non-empty strings") + if not isinstance(value["acceptance"], list) or not value["acceptance"]: + raise ContractError("acceptance must be non-empty") + for item in value["acceptance"]: + _exact(item, {"id", "description", "evidence_kind"}, {"id", "description", "evidence_kind"}, "acceptance item") + if item["evidence_kind"] not in {"review", "check", "artifact", "host"}: + raise ContractError("invalid evidence_kind") + if not isinstance(value["checks"], list): raise ContractError("checks must be an array") + for check in value["checks"]: + _exact(check, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, "check") + if not isinstance(check["argv"], list) or not check["argv"] or not all(isinstance(v, str) and v for v in check["argv"]): raise ContractError("check argv must be a non-empty string array") + _relative(check["cwd"], "check cwd") + if not isinstance(check["timeout_seconds"], int) or isinstance(check["timeout_seconds"], bool) or check["timeout_seconds"] <= 0: raise ContractError("check timeout must be positive") + scope = value["scope"]; _exact(scope, {"read_paths", "write_paths"}, {"read_paths", "write_paths"}, "scope") + for p in scope["read_paths"] + scope["write_paths"]: _relative(p, "scope path") + if value["workflow"] == "branch-review" and scope["write_paths"]: raise ContractError("branch review cannot write") + budget = value["budget"]; required = {"wall_seconds", "max_worker_invocations", "max_revisions", "max_fallbacks_per_step"}; _exact(budget, required, required, "budget") + for key, number in budget.items(): + if not isinstance(number, int) or isinstance(number, bool) or number < 0: raise ContractError(f"budget {key} must be a finite non-negative integer") + if budget["wall_seconds"] == 0 or budget["max_worker_invocations"] == 0: raise ContractError("wall_seconds and max_worker_invocations must be positive") diff --git a/plugin/core/profiles/templates.json b/plugin/core/profiles/templates.json new file mode 100644 index 0000000..7840853 --- /dev/null +++ b/plugin/core/profiles/templates.json @@ -0,0 +1,9 @@ +{ + "schema_version": 1, + "templates": { + "implement.balanced": {"permissions": "workspace_write", "allowed_tools": ["read", "edit", "shell"], "billing": "subscription_only"}, + "review.deep": {"permissions": "read_only", "allowed_tools": ["read", "shell"], "billing": "subscription_only"}, + "research.current": {"permissions": "read_only", "allowed_tools": ["read", "web"], "billing": "subscription_only"} + }, + "bindings": {} +} diff --git a/plugin/core/pyproject.toml b/plugin/core/pyproject.toml new file mode 100644 index 0000000..ef8f93b --- /dev/null +++ b/plugin/core/pyproject.toml @@ -0,0 +1,25 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "devsquad-core" +version = "0.1.0" +requires-python = ">=3.11" +dependencies = [] + +[project.scripts] +squad = "devsquad.cli:main" + +[tool.setuptools] +package-dir = {"" = "src"} + +[tool.setuptools.packages.find] +where = ["src"] + +[tool.setuptools.data-files] +"share/devsquad/adapters/codex" = ["adapters/codex/adapter.json"] +"share/devsquad/adapters/antigravity" = ["adapters/antigravity/adapter.json"] +"share/devsquad/adapters/grok" = ["adapters/grok/adapter.json"] +"share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/task.schema.json"] +"share/devsquad/profiles" = ["profiles/templates.json"] diff --git a/plugin/core/schemas/adapter.schema.json b/plugin/core/schemas/adapter.schema.json new file mode 100644 index 0000000..b8fb04b --- /dev/null +++ b/plugin/core/schemas/adapter.schema.json @@ -0,0 +1,15 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/adapter-v1.json", + "type": "object", + "additionalProperties": true, + "required": ["schema_version", "name", "transport", "binary_candidates", "capabilities", "permission_profiles"], + "properties": { + "schema_version": {"const": 1}, + "name": {"type": "string", "minLength": 1}, + "transport": {"enum": ["cli_exec", "native_protocol"]}, + "binary_candidates": {"type": "array", "minItems": 1, "items": {"type": "string"}}, + "capabilities": {"type": "object"}, + "permission_profiles": {"type": "object"} + } +} diff --git a/plugin/core/schemas/execution-identity.schema.json b/plugin/core/schemas/execution-identity.schema.json new file mode 100644 index 0000000..5b248ed --- /dev/null +++ b/plugin/core/schemas/execution-identity.schema.json @@ -0,0 +1,13 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/execution-identity-v1.json", + "type": "object", "additionalProperties": false, + "required": ["harness", "harness_version", "model_provider", "model_family", "model", "effort", "tools", "permissions", "account_pool", "verification"], + "properties": { + "harness": {"type": "string", "minLength": 1}, "harness_version": {"type": ["string", "null"]}, + "model_provider": {"type": ["string", "null"]}, "model_family": {"type": ["string", "null"]}, "model": {"type": ["string", "null"]}, + "effort": {"type": ["string", "null"]}, "tools": {"type": "array", "items": {"type": "string"}, "uniqueItems": true}, + "permissions": {"enum": ["read_only", "workspace_write"]}, "account_pool": {"type": ["string", "null"]}, + "verification": {"enum": ["verified", "unverified", "unavailable", "unknown"]} + } +} diff --git a/plugin/core/schemas/launch-spec.schema.json b/plugin/core/schemas/launch-spec.schema.json new file mode 100644 index 0000000..aa6d282 --- /dev/null +++ b/plugin/core/schemas/launch-spec.schema.json @@ -0,0 +1,18 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/launch-spec-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "adapter", "transport", "argv", "cwd", "stdin_path", "timeout_seconds", "requested", "environment"], + "properties": { + "schema_version": {"const": 1}, + "adapter": {"type": "string", "minLength": 1}, + "transport": {"enum": ["cli_exec", "native_protocol"]}, + "argv": {"type": "array", "minItems": 1, "items": {"type": "string", "minLength": 1}}, + "cwd": {"type": "string", "minLength": 1}, + "stdin_path": {"type": ["string", "null"]}, + "timeout_seconds": {"type": "integer", "minimum": 1}, + "requested": {"$ref": "execution-identity.schema.json"}, + "environment": {"type": "object", "additionalProperties": {"type": "string"}} + } +} diff --git a/plugin/core/schemas/normalized-result.schema.json b/plugin/core/schemas/normalized-result.schema.json new file mode 100644 index 0000000..e7c4f71 --- /dev/null +++ b/plugin/core/schemas/normalized-result.schema.json @@ -0,0 +1,19 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/normalized-result-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "execution_status", "error_code", "output", "artifact_status", "acceptance_status", "requested", "observed", "native_ids", "events"], + "properties": { + "schema_version": {"const": 1}, + "execution_status": {"enum": ["succeeded", "failed", "timed_out", "interrupted", "denied", "malformed"]}, + "error_code": {"type": ["string", "null"]}, + "output": {"type": ["string", "null"]}, + "artifact_status": {"enum": ["present", "missing", "not_required", "unknown"]}, + "acceptance_status": {"enum": ["pending", "accepted", "rejected", "not_evaluated"]}, + "requested": {"$ref": "execution-identity.schema.json"}, + "observed": {"oneOf": [{"$ref": "execution-identity.schema.json"}, {"type": "null"}]}, + "native_ids": {"type": "object", "additionalProperties": {"type": "string"}}, + "events": {"type": "array", "items": {"type": "object"}} + } +} diff --git a/plugin/core/schemas/policy.schema.json b/plugin/core/schemas/policy.schema.json new file mode 100644 index 0000000..9541070 --- /dev/null +++ b/plugin/core/schemas/policy.schema.json @@ -0,0 +1,6 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", "$id": "https://devsquad.local/schemas/policy-v1.json", + "type": "object", "additionalProperties": false, + "required": ["schema_version", "id", "version", "roles", "task_classes", "require_different_model_for_review", "account_pools", "experiment_budget"], + "properties": {"schema_version": {"const": 1}, "id": {"type": "string", "minLength": 1}, "version": {"type": "integer", "minimum": 1}, "roles": {"type": "object"}, "task_classes": {"type": "object"}, "require_different_model_for_review": {"type": "boolean"}, "prefer_different_harness_for_review": {"type": "boolean"}, "account_pools": {"type": "object"}, "experiment_budget": {"type": "object"}} +} diff --git a/plugin/core/schemas/profile.schema.json b/plugin/core/schemas/profile.schema.json new file mode 100644 index 0000000..2c874a0 --- /dev/null +++ b/plugin/core/schemas/profile.schema.json @@ -0,0 +1,20 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/profile-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["id", "harness", "model_family", "model_id", "effort", "required_tools", "permission_policy", "account_pool_id", "billing_mode", "quality_status", "evidence_refs"], + "properties": { + "id": {"type": "string", "minLength": 1}, + "harness": {"type": "string", "minLength": 1}, + "model_family": {"type": "string", "minLength": 1}, + "model_id": {"type": "string", "minLength": 1}, + "effort": {"type": "object", "additionalProperties": false, "required": ["value", "transport"], "properties": {"value": {"type": ["string", "null"]}, "transport": {"enum": ["native", "model_variant", "provider_default"]}}}, + "required_tools": {"type": "array", "items": {"type": "string"}, "uniqueItems": true}, + "permission_policy": {"enum": ["read_only", "workspace_write"]}, + "account_pool_id": {"type": "string", "minLength": 1}, + "billing_mode": {"enum": ["subscription", "paid_api"]}, + "quality_status": {"enum": ["unvalidated", "trial", "proven", "suspended"]}, + "evidence_refs": {"type": "array", "items": {"type": "string"}, "uniqueItems": true} + } +} diff --git a/plugin/core/schemas/task.schema.json b/plugin/core/schemas/task.schema.json new file mode 100644 index 0000000..01207f5 --- /dev/null +++ b/plugin/core/schemas/task.schema.json @@ -0,0 +1,11 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", "$id": "https://devsquad.local/schemas/task-v1.json", + "type": "object", "additionalProperties": false, + "required": ["schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin"], + "properties": { + "schema_version": {"const": 1}, + "project": {"type": "object", "additionalProperties": false, "required": ["repo_path", "base_ref", "target_ref"], "properties": {"repo_path": {"type": "string"}, "base_ref": {"type": "string"}, "target_ref": {"type": "string"}}}, + "workflow": {"enum": ["branch-review", "issue-delivery"]}, "goal": {"type": "string", "minLength": 1}, "task_class": {"type": "string", "minLength": 1}, + "acceptance": {"type": "array", "minItems": 1}, "checks": {"type": "array"}, "scope": {"type": "object"}, "lead": {"type": "object"}, "routing": {"type": "object"}, "budget": {"type": "object"}, "origin": {"type": "object"} + } +} diff --git a/plugin/core/src/devsquad/__init__.py b/plugin/core/src/devsquad/__init__.py new file mode 100644 index 0000000..22a4c4a --- /dev/null +++ b/plugin/core/src/devsquad/__init__.py @@ -0,0 +1,3 @@ +"""DevSquad's surface-independent local core.""" + +__version__ = "0.1.0" diff --git a/plugin/core/src/devsquad/adapters.py b/plugin/core/src/devsquad/adapters.py new file mode 100644 index 0000000..70d2bcd --- /dev/null +++ b/plugin/core/src/devsquad/adapters.py @@ -0,0 +1,161 @@ +"""Manifest-driven M1 adapter preparation and output classification.""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import subprocess +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from .contracts import ContractError, ExecutionIdentity, LaunchSpec, NormalizedResult, ProfileUnsupported, SCHEMA_VERSION + +ERROR_PATTERNS = ( + ("AUTH_ERROR", re.compile(r"auth|unauthorized|ineligible|\b401\b|\b403\b", re.I)), + ("RATE_LIMITED", re.compile(r"rate.?limit|quota|resource.?exhausted|too many requests|\b429\b", re.I)), +) + + +@dataclass(frozen=True) +class AdapterManifest: + name: str + transport: str + binary_candidates: tuple[str, ...] + model_provider: str | None + efforts_by_model: dict[str, tuple[str, ...]] + permission_profiles: dict[str, tuple[str, ...]] + output_format: str + verified_versions: tuple[str, ...] + + @classmethod + def load(cls, path: Path) -> "AdapterManifest": + raw = json.loads(path.read_text()) + required = {"schema_version", "name", "transport", "binary_candidates", "capabilities"} + missing = required - raw.keys() + if missing or raw["schema_version"] != 1: + raise ContractError(f"invalid adapter manifest: missing={sorted(missing)}") + capabilities = raw["capabilities"] + return cls( + name=raw["name"], transport=raw["transport"], + binary_candidates=tuple(raw["binary_candidates"]), + model_provider=raw.get("model_provider"), + efforts_by_model={k: tuple(v) for k, v in capabilities.get("efforts_by_model", {}).items()}, + permission_profiles={k: tuple(v) for k, v in raw.get("permission_profiles", {}).items()}, + output_format=raw.get("output_format", "text"), + verified_versions=tuple(raw.get("verified_harness_versions", [])), + ) + + def resolve_binary(self) -> str | None: + return next((p for name in self.binary_candidates if (p := shutil.which(name))), None) + + +def _permission_args(manifest: AdapterManifest, permission: str) -> tuple[str, ...]: + try: + return manifest.permission_profiles[permission] + except KeyError as exc: + raise ContractError(f"unsupported permission profile: {permission}") from exc + + +def prepare_cli( + manifest: AdapterManifest, *, prompt: str, cwd: str, model: str | None, + effort: str | None, permission: str, timeout_seconds: int, stdin_path: str | None = None, +) -> LaunchSpec: + binary = manifest.resolve_binary() + if not binary: + raise ContractError(f"adapter unavailable: {manifest.name}") + if effort is not None: + supported = manifest.efforts_by_model.get(model or "") + if supported is None or effort not in supported: + raise ProfileUnsupported(f"unsupported or unverified effort {effort!r} for {manifest.name} model {model!r}") + args: list[str] + if manifest.name == "codex": + sandbox = "read-only" if permission == "read_only" else "workspace-write" + args = [binary, "exec", "--json", "--cd", cwd, "--sandbox", sandbox] + if model: + args += ["--model", model] + if effort: + args += ["-c", f'model_reasoning_effort="{effort}"'] + args += [prompt] + elif manifest.name == "antigravity": + args = [binary, "--print", prompt, "--output-format", "json", "--mode", "plan" if permission == "read_only" else "accept-edits"] + if model: + args += ["--model", model] + if effort: + args += ["--effort", effort] + elif manifest.name == "grok": + args = [binary, "--single", prompt, "--output-format", "json", "--cwd", cwd, "--permission-mode", "plan" if permission == "read_only" else "acceptEdits", "--no-subagents"] + if model: + args += ["--model", model] + if effort: + args += ["--reasoning-effort", effort] + else: + raise ContractError(f"no argv builder for adapter: {manifest.name}") + args.extend(_permission_args(manifest, permission)) + requested = ExecutionIdentity( + harness=manifest.name, harness_version=None, model_provider=manifest.model_provider, + model_family=None, model=model, effort=effort, permissions=permission, + verification="unverified", + ) + return LaunchSpec(SCHEMA_VERSION, manifest.name, "cli_exec", tuple(args), str(Path(cwd).resolve()), stdin_path, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) + + +def _provider_records(adapter: str, stdout: str) -> tuple[list[dict[str, Any]], bool, str | None]: + records: list[dict[str, Any]] = [] + deliverable = False + error_text = None + try: + document = json.loads(stdout) + source = document if isinstance(document, list) else [document] + except json.JSONDecodeError: + source = [json.loads(line) for line in stdout.splitlines() if line.strip()] + for item in source: + if not isinstance(item, dict): + raise json.JSONDecodeError("record is not an object", line, 0) + records.append(item) + if item.get("is_error") is True or item.get("error"): + error_text = str(item.get("error") or item.get("result") or item.get("message")) + kind = item.get("type") + if adapter == "codex" and kind == "item.completed": + native_item = item.get("item") or {} + if native_item.get("type") in {"agent_message", "agentMessage"}: + payload = native_item.get("text") or native_item.get("content") + deliverable = isinstance(payload, str) and bool(payload.strip()) + if kind in {"result", "assistant_message", "turn.completed"} and item.get("is_error") is not True: + payload = item.get("result") or item.get("message") or item.get("text") + if isinstance(payload, str) and payload.strip(): + deliverable = True + return records, deliverable, error_text + + +def classify_cli(spec: LaunchSpec, *, returncode: int, stdout: str, stderr: str, timed_out: bool = False) -> NormalizedResult: + code = next((code for code, pattern in ERROR_PATTERNS if pattern.search(stderr)), None) + status = "succeeded" + if code == "AUTH_ERROR" or code == "RATE_LIMITED": + status = "failed" + elif timed_out or returncode in (124, 137, 143): + status, code = "timed_out", "TIMEOUT" + elif returncode != 0: + status, code = "failed", "CLI_ERROR" + elif not stdout.strip(): + status, code = "malformed", "CLI_ERROR" + elif spec.adapter in {"codex", "antigravity", "grok"}: + try: + _, deliverable, provider_error = _provider_records(spec.adapter, stdout) + if provider_error: + code = next((candidate for candidate, pattern in ERROR_PATTERNS if pattern.search(provider_error)), "CLI_ERROR") + status = "denied" if re.search(r"permission denied|tool (?:use )?denied|not allowed", provider_error, re.I) else "failed" + elif not deliverable: + status, code = "malformed", "CLI_ERROR" + except json.JSONDecodeError: + status, code = "malformed", "CLI_ERROR" + return NormalizedResult(SCHEMA_VERSION, status, code, stdout if stdout else None, "unknown", "not_evaluated", spec.requested, None) + + +def harness_version(binary: str) -> str | None: + try: + return subprocess.run([binary, "--version"], text=True, capture_output=True, timeout=3, check=False).stdout.strip() or None + except (OSError, subprocess.TimeoutExpired): + return None diff --git a/plugin/core/src/devsquad/catalog.py b/plugin/core/src/devsquad/catalog.py new file mode 100644 index 0000000..59bc94c --- /dev/null +++ b/plugin/core/src/devsquad/catalog.py @@ -0,0 +1,60 @@ +"""Structured, last-good model catalog handling for M1.""" + +from __future__ import annotations + +import hashlib +import json +import os +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable + +from .contracts import ContractError + + +def model_fingerprint(harness: str, version: str | None, model: dict[str, Any]) -> str: + stable = {"harness": harness, "version": version, "model": model} + return hashlib.sha256(json.dumps(stable, sort_keys=True, separators=(",", ":")).encode()).hexdigest() + + +def normalize_models(harness: str, version: str | None, models: Iterable[dict[str, Any]]) -> list[dict[str, Any]]: + normalized = [] + for raw in models: + model_id = raw.get("id") or raw.get("model") + if not isinstance(model_id, str) or not model_id: + raise ContractError("catalog model missing id") + effort_values = raw.get("supportedReasoningEfforts") or raw.get("supported_reasoning_efforts") or [] + efforts = [item.get("reasoningEffort") if isinstance(item, dict) else item for item in effort_values] + normalized.append({ + "id": model_id, + "display_name": raw.get("displayName") or raw.get("display_name") or model_id, + "family": raw.get("family"), + "default_effort": raw.get("defaultReasoningEffort") or raw.get("default_reasoning_effort"), + "supported_efforts": efforts, + "modalities": raw.get("inputModalities") or raw.get("input_modalities") or [], + "is_default": bool(raw.get("isDefault") or raw.get("is_default")), + "qualification": "unqualified", + "fingerprint": model_fingerprint(harness, version, raw), + }) + return normalized + + +def update_last_good(path: Path, *, harness: str, version: str | None, models: Iterable[dict[str, Any]] | None, complete: bool, error: str | None = None) -> dict[str, Any]: + old = json.loads(path.read_text()) if path.exists() else None + now = datetime.now(timezone.utc).isoformat() + if error or not complete or models is None: + if old: + old["last_refresh"] = {"at": now, "status": "error" if error else "incomplete", "error": error} + _atomic_json(path, old) + return old + raise ContractError(error or "catalog response incomplete and no last-good snapshot exists") + value = {"schema_version": 1, "harness": harness, "harness_version": version, "fetched_at": now, "complete": True, "models": normalize_models(harness, version, models), "last_refresh": {"at": now, "status": "ok", "error": None}} + _atomic_json(path, value) + return value + + +def _atomic_json(path: Path, value: dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + tmp = path.with_name(f"{path.name}.tmp.{os.getpid()}") + tmp.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") + os.replace(tmp, path) diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py new file mode 100644 index 0000000..565c647 --- /dev/null +++ b/plugin/core/src/devsquad/cli.py @@ -0,0 +1,87 @@ +"""Small M1 command surface: version, doctor, prepare and classify.""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +from . import __version__ +from .adapters import AdapterManifest, classify_cli, harness_version, prepare_cli +from .contracts import ContractError, envelope, error_payload + +SOURCE_ROOT = Path(__file__).resolve().parents[2] +CORE_ROOT = SOURCE_ROOT if (SOURCE_ROOT / "adapters").is_dir() else Path(sys.prefix) / "share" / "devsquad" + + +def manifests() -> list[tuple[Path, AdapterManifest]]: + return [(path, AdapterManifest.load(path)) for path in sorted((CORE_ROOT / "adapters").glob("*/adapter.json"))] + + +def command_doctor(_: argparse.Namespace) -> tuple[dict, int]: + rows = [] + for path, manifest in manifests(): + binary = manifest.resolve_binary() + version = harness_version(binary) if binary else None + status = "unavailable" if not binary else ("supported" if version in manifest.verified_versions else "unverified") + rows.append({"adapter": manifest.name, "transport": manifest.transport, "status": status, "binary": binary, "version": version, "manifest": str(path)}) + ready = any(row["status"] != "unavailable" for row in rows) + return envelope(data={"core_version": __version__, "ready": ready, "adapters": rows}), 0 if ready else 1 + + +def command_prepare(args: argparse.Namespace) -> dict: + manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") + spec = prepare_cli(manifest, prompt=args.prompt, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) + return envelope(data=spec.to_dict()), 0 + + +def command_classify(args: argparse.Namespace) -> dict: + manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") + spec = prepare_cli(manifest, prompt="classification", cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) + result = classify_cli(spec, returncode=args.returncode, stdout=Path(args.stdout_file).read_text(), stderr=Path(args.stderr_file).read_text()) + return envelope(data=result.to_dict()), 0 + + +class ContractParser(argparse.ArgumentParser): + def error(self, message: str) -> None: + raise ContractError(message) + + +def parser() -> argparse.ArgumentParser: + p = ContractParser(prog="squad") + p.add_argument("--version", action="version", version=f"squad {__version__}") + sub = p.add_subparsers(dest="command", required=True) + doctor = sub.add_parser("doctor"); doctor.add_argument("--json", action="store_true"); doctor.set_defaults(func=command_doctor) + for name, fn in (("prepare", command_prepare), ("classify", command_classify)): + cmd = sub.add_parser(name) + cmd.add_argument("adapter", choices=("codex", "antigravity", "grok")) + cmd.add_argument("--cwd", default=str(Path.cwd())) + cmd.add_argument("--model") + cmd.add_argument("--effort") + cmd.add_argument("--permission", choices=("read_only", "workspace_write"), default="read_only") + cmd.add_argument("--timeout", type=int, default=90) + if name == "prepare": + cmd.add_argument("--prompt", required=True) + else: + cmd.add_argument("--returncode", type=int, required=True) + cmd.add_argument("--stdout-file", required=True) + cmd.add_argument("--stderr-file", required=True) + cmd.set_defaults(func=fn) + return p + + +def main(argv: list[str] | None = None) -> int: + try: + args = parser().parse_args(argv) + response, code = args.func(args) + print(json.dumps(response, sort_keys=True)) + return code + except (ContractError, OSError, json.JSONDecodeError) as exc: + code = getattr(exc, "code", "INPUT_INVALID") + print(json.dumps(envelope(error=error_payload(code, str(exc))), sort_keys=True)) + return 64 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py new file mode 100644 index 0000000..edfc56e --- /dev/null +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -0,0 +1,76 @@ +"""Typed JSON-RPC preparation/state normalization for Codex app-server. + +This module does not spawn or supervise the server. M2 gives it a connected +stdio stream owned by the run supervisor. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any + +from .contracts import ContractError + + +def request(request_id: int, method: str, params: dict[str, Any] | None = None) -> dict[str, Any]: + return {"id": request_id, "method": method, "params": params or {}} + + +def initialize_request(request_id: int = 1) -> dict[str, Any]: + return request(request_id, "initialize", {"clientInfo": {"name": "devsquad", "version": "0.1.0"}, "capabilities": {"experimentalApi": True}}) + + +def model_list_request(request_id: int, cursor: str | None = None, limit: int = 100) -> dict[str, Any]: + params: dict[str, Any] = {"limit": limit} + if cursor: + params["cursor"] = cursor + return request(request_id, "model/list", params) + + +def turn_start_request(request_id: int, *, thread_id: str, prompt: str, model: str | None, effort: str | None, cwd: str, sandbox: str) -> dict[str, Any]: + params: dict[str, Any] = {"threadId": thread_id, "input": [{"type": "text", "text": prompt}], "cwd": cwd, "permissions": sandbox} + if model: + params["model"] = model + if effort: + params["effort"] = effort + return request(request_id, "turn/start", params) + + +@dataclass +class NativeTurnState: + thread_id: str | None = None + turn_id: str | None = None + terminal: bool = False + interrupted_acknowledged: bool = False + output: list[str] = field(default_factory=list) + events: list[dict[str, Any]] = field(default_factory=list) + + def consume(self, message: dict[str, Any]) -> None: + if not isinstance(message, dict): + raise ContractError("native message must be an object") + self.events.append(message) + method = message.get("method", "") + params = message.get("params") or message.get("result") or {} + if method in {"thread/started", "thread/start/completed"}: + self.thread_id = params.get("thread", {}).get("id") or params.get("threadId") or self.thread_id + if method in {"turn/started", "turn/start/completed"}: + self.turn_id = params.get("turn", {}).get("id") or params.get("turnId") or self.turn_id + if method in {"item/agentMessage/delta", "turn/output/delta"}: + self.output.append(params.get("delta", "")) + if method in {"turn/interrupt/completed", "turn/interrupted/acknowledged"}: + self.interrupted_acknowledged = True + if method in {"turn/completed", "review/completed", "turn/failed", "turn/interrupted"}: + self.terminal = True + + +def parse_model_page(response: dict[str, Any]) -> tuple[list[dict[str, Any]], str | None]: + if "error" in response: + raise ContractError(f"model/list failed: {response['error']}") + result = response.get("result") + if not isinstance(result, dict): + raise ContractError("model/list missing result") + models = result.get("data") or result.get("models") + if not isinstance(models, list): + raise ContractError("model/list is incomplete") + cursor = result.get("nextCursor") or result.get("next_cursor") + return models, cursor diff --git a/plugin/core/src/devsquad/contracts.py b/plugin/core/src/devsquad/contracts.py new file mode 100644 index 0000000..9fca7dc --- /dev/null +++ b/plugin/core/src/devsquad/contracts.py @@ -0,0 +1,103 @@ +"""Strict M1 contracts for prepared invocations and normalized results.""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Any, Literal + +SCHEMA_VERSION = 1 +Transport = Literal["cli_exec", "native_protocol"] +Verification = Literal["verified", "unverified", "unavailable", "unknown"] + + +class ContractError(ValueError): + """Raised before launch when an invocation contract is unsupported.""" + + code = "INPUT_INVALID" + + +class ProfileUnsupported(ContractError): + code = "PROFILE_UNSUPPORTED" + + +@dataclass(frozen=True) +class ExecutionIdentity: + harness: str + harness_version: str | None + model_provider: str | None + model_family: str | None + model: str | None + effort: str | None + tools: tuple[str, ...] = () + permissions: str = "read_only" + account_pool: str | None = None + verification: Verification = "unknown" + + +@dataclass(frozen=True) +class LaunchSpec: + """A launch description. M2 owns spawning, timeout, cancellation and reaping.""" + + schema_version: int + adapter: str + transport: Transport + argv: tuple[str, ...] + cwd: str + stdin_path: str | None + timeout_seconds: int + requested: ExecutionIdentity + environment: dict[str, str] = field(default_factory=dict) + + def __post_init__(self) -> None: + if self.schema_version != SCHEMA_VERSION: + raise ContractError("unsupported schema_version") + if self.transport not in ("cli_exec", "native_protocol"): + raise ContractError("unsupported transport") + if not self.argv or not all(isinstance(v, str) and v for v in self.argv): + raise ContractError("argv must be a non-empty string array") + if self.timeout_seconds <= 0: + raise ContractError("timeout_seconds must be positive") + if not Path(self.cwd).is_absolute(): + raise ContractError("cwd must be absolute") + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +@dataclass(frozen=True) +class NormalizedResult: + """Execution evidence without conflating artifacts or acceptance.""" + + schema_version: int + execution_status: Literal[ + "succeeded", "failed", "timed_out", "interrupted", "denied", "malformed" + ] + error_code: str | None + output: str | None + artifact_status: Literal["present", "missing", "not_required", "unknown"] + acceptance_status: Literal["pending", "accepted", "rejected", "not_evaluated"] + requested: ExecutionIdentity + observed: ExecutionIdentity | None + native_ids: dict[str, str] = field(default_factory=dict) + events: tuple[dict[str, Any], ...] = () + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + +def envelope(*, data: Any = None, error: dict[str, Any] | None = None) -> dict[str, Any]: + if (data is None) == (error is None): + raise ContractError("exactly one of data and error is required") + return {"schema_version": SCHEMA_VERSION, "ok": error is None, "data": data, "error": error} + + +def error_payload(code: str, message: str, *, retryable: bool = False, details: dict[str, Any] | None = None) -> dict[str, Any]: + return {"code": code, "message": message, "retryable": retryable, "details": details or {}} + + +def validate_launch_payload(value: dict[str, Any]) -> None: + expected = {"schema_version", "adapter", "transport", "argv", "cwd", "stdin_path", "timeout_seconds", "requested", "environment"} + if set(value) != expected: + raise ContractError(f"LaunchSpec fields differ: {sorted(set(value) ^ expected)}") + LaunchSpec(**{**value, "argv": tuple(value["argv"]), "requested": ExecutionIdentity(**{**value["requested"], "tools": tuple(value["requested"]["tools"])})}) diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py new file mode 100644 index 0000000..3be2dc9 --- /dev/null +++ b/plugin/core/src/devsquad/validation.py @@ -0,0 +1,66 @@ +"""Dependency-free strict validation for M1 public fixtures.""" + +from __future__ import annotations +from pathlib import Path +from typing import Any +from .contracts import ContractError + +TASK_FIELDS = {"schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin", "review"} + + +def _exact(value: dict[str, Any], allowed: set[str], required: set[str], label: str) -> None: + if not isinstance(value, dict): + raise ContractError(f"{label} must be an object") + unknown, missing = set(value) - allowed, required - set(value) + if unknown or missing: + raise ContractError(f"{label} fields invalid: unknown={sorted(unknown)} missing={sorted(missing)}") + + +def _relative(path: str, label: str) -> None: + p = Path(path) + if p.is_absolute() or ".." in p.parts: + raise ContractError(f"{label} must be repository-relative without traversal") + + +def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) -> None: + _exact(value, TASK_FIELDS, TASK_FIELDS - {"review"}, "task") + if type(value["schema_version"]) is not int or value["schema_version"] != 1 or value["workflow"] not in {"branch-review", "issue-delivery"}: + raise ContractError("unsupported task schema or workflow") + project = value["project"] + _exact(project, {"repo_path", "base_ref", "target_ref"}, {"repo_path", "base_ref", "target_ref"}, "project") + if not Path(project["repo_path"]).is_absolute() or (require_existing_repo and not (Path(project["repo_path"]) / ".git").exists()): + raise ContractError("project.repo_path must be an existing absolute Git repository") + if not isinstance(value["goal"], str) or not value["goal"].strip() or not isinstance(value["task_class"], str) or not value["task_class"].strip(): + raise ContractError("goal and task_class must be non-empty strings") + if not isinstance(value["acceptance"], list) or not value["acceptance"]: + raise ContractError("acceptance must be non-empty") + for item in value["acceptance"]: + _exact(item, {"id", "description", "evidence_kind"}, {"id", "description", "evidence_kind"}, "acceptance item") + if item["evidence_kind"] not in {"review", "check", "artifact", "host"}: + raise ContractError("invalid evidence_kind") + if not isinstance(value["checks"], list): raise ContractError("checks must be an array") + for check in value["checks"]: + _exact(check, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, "check") + if not isinstance(check["argv"], list) or not check["argv"] or not all(isinstance(v, str) and v for v in check["argv"]): raise ContractError("check argv must be a non-empty string array") + _relative(check["cwd"], "check cwd") + if not isinstance(check["timeout_seconds"], int) or isinstance(check["timeout_seconds"], bool) or check["timeout_seconds"] <= 0: raise ContractError("check timeout must be positive") + if type(check["required_to_pass"]) is not bool: raise ContractError("required_to_pass must be boolean") + scope = value["scope"]; _exact(scope, {"read_paths", "write_paths"}, {"read_paths", "write_paths"}, "scope") + if not isinstance(scope["read_paths"], list) or not isinstance(scope["write_paths"], list) or not all(isinstance(p, str) for p in scope["read_paths"] + scope["write_paths"]): raise ContractError("scope paths must be string arrays") + for p in scope["read_paths"] + scope["write_paths"]: _relative(p, "scope path") + if value["workflow"] == "branch-review" and scope["write_paths"]: raise ContractError("branch review cannot write") + lead = value["lead"]; _exact(lead, {"mode"}, {"mode"}, "lead") + if lead["mode"] not in {"host", "headless"}: raise ContractError("invalid lead mode") + routing = value["routing"]; _exact(routing, {"profiles_file", "policy_file", "overrides"}, {"profiles_file", "policy_file"}, "routing") + for key in ("profiles_file", "policy_file"): + if not isinstance(routing[key], str) or not routing[key]: raise ContractError(f"routing {key} must be a path") + origin = value["origin"]; _exact(origin, {"surface", "session_ref"}, {"surface"}, "origin") + if not isinstance(origin["surface"], str) or not origin["surface"]: raise ContractError("origin surface must be non-empty") + if "review" in value: + review = value["review"]; _exact(review, {"mode", "focus"}, {"mode"}, "review") + if review["mode"] not in {"standard", "adversarial"}: raise ContractError("invalid review mode") + if "focus" in review and review["mode"] != "adversarial": raise ContractError("review focus requires adversarial mode") + budget = value["budget"]; required = {"wall_seconds", "max_worker_invocations", "max_revisions", "max_fallbacks_per_step"}; _exact(budget, required, required, "budget") + for key, number in budget.items(): + if not isinstance(number, int) or isinstance(number, bool) or number < 0: raise ContractError(f"budget {key} must be a finite non-negative integer") + if budget["wall_seconds"] == 0 or budget["max_worker_invocations"] == 0: raise ContractError("wall_seconds and max_worker_invocations must be positive") diff --git a/plugin/lib/adapter.sh b/plugin/lib/adapter.sh index 989ae16..6679766 100644 --- a/plugin/lib/adapter.sh +++ b/plugin/lib/adapter.sh @@ -33,6 +33,26 @@ _ADAPTER_LIB_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # shellcheck source=/dev/null source "${_ADAPTER_LIB_DIR}/model-catalog.sh" +# Terminate a bounded subprocess tree without requiring GNU timeout, setsid, +# or job-control process groups (all absent on a stock macOS Bash 3.2 host). +# Descendants are collected before the parent so an exiting parent cannot +# orphan children between discovery and signalling. +_adapter_snapshot_tree() { + local root_pid="$1" child + for child in $(pgrep -P "$root_pid" 2>/dev/null || true); do + _adapter_snapshot_tree "$child" + done + printf '%s\n' "$root_pid" +} + +_adapter_signal_snapshot() { + local snapshot_file="$1" signal="${2:-TERM}" pid + [[ -f "$snapshot_file" ]] || return 0 + while IFS= read -r pid; do + [[ -n "$pid" ]] && kill -"$signal" "$pid" 2>/dev/null || true + done < "$snapshot_file" +} + # Resolve model: agent-specific (agent_models.) > # global (.preferences.) > "" (CLI default). # Values may be exact model names OR tiers ("tier:fast" / "tier:frontier"), @@ -135,7 +155,11 @@ _adapter_invoke() { "$cli" "${ADAPTER_ARGS[@]}" >"$stdout_file" 2>"$stderr_file" & fi local cli_pid=$! - ( sleep "$timeout_secs"; kill "$cli_pid" 2>/dev/null ) & + local timed_out_file="${stdout_file}.timed-out" + local process_snapshot="${stdout_file}.processes" + # Redirect the watchdog itself: inherited capture descriptors were the + # reason a successful immediate command waited for the full timeout. + ( sleep "$timeout_secs"; _adapter_snapshot_tree "$cli_pid" > "$process_snapshot"; : > "$timed_out_file"; _adapter_signal_snapshot "$process_snapshot" TERM ) >/dev/null 2>&1 & local watchdog_pid=$! if wait "$cli_pid"; then exit_code=0 @@ -144,10 +168,13 @@ _adapter_invoke() { fi kill "$watchdog_pid" 2>/dev/null || true wait "$watchdog_pid" 2>/dev/null || true - # SIGTERM from the watchdog surfaces as 143 — normalize to timeout's 124 - if [[ $exit_code -eq 143 ]]; then + if [[ -f "$timed_out_file" ]]; then + # Give descendants a brief grace, then ensure stubborn children vanish. + sleep 0.1 + _adapter_signal_snapshot "$process_snapshot" KILL exit_code=124 fi + rm -f "$timed_out_file" "$process_snapshot" fi local stdout stderr_content diff --git a/plugin/lib/gemini-wrapper.sh b/plugin/lib/gemini-wrapper.sh index 46a1e1f..f772c02 100755 --- a/plugin/lib/gemini-wrapper.sh +++ b/plugin/lib/gemini-wrapper.sh @@ -104,9 +104,10 @@ invoke_gemini() { _adapter_invoke "$final_prompt" "$timeout_secs" } -# File-based invocation: concatenates file/dir contents and pipes them via -# stdin (bypasses Antigravity's workspace sandbox restriction on @file refs). -# Usage: invoke_gemini_with_files "@src/auth/ @src/models/user.ts" "prompt" [word_limit] [timeout_secs] +# File-based invocation. Context arguments are newline-delimited so paths with +# spaces remain intact. A single legacy whitespace-delimited argument remains +# accepted only when every token resolves, preserving existing callers. +# Usage: invoke_gemini_with_files $'@src/auth/\n@src/models/user.ts' "prompt" [word_limit] [timeout_secs] invoke_gemini_with_files() { local files_arg="$1" local prompt="$2" @@ -117,22 +118,71 @@ invoke_gemini_with_files() { # expand backslash escapes INSIDE file contents, corrupting code) local nl=$'\n' local file_content="" - local token path f - for token in $files_arg; do + local token path f manifest="$files_arg" + local project_root="${CLAUDE_PROJECT_DIR:-.}" + local max_bytes="${DEVSQUAD_CONTEXT_MAX_BYTES:-1048576}" + local max_file_bytes="${DEVSQUAD_CONTEXT_MAX_FILE_BYTES:-262144}" + local used_bytes=0 file_bytes header_bytes + project_root=$(cd "$project_root" 2>/dev/null && pwd -P) || { + echo "CONTEXT_OMITTED: project root is unavailable: ${project_root}" >&2 + return 1 + } + if [[ "$files_arg" != *$'\n'* ]]; then + local all_tokens_resolve="true" + for token in $files_arg; do + path="${token#@}" + [[ "$token" == @* && ( -f "$project_root/$path" || -d "$project_root/$path" ) ]] || all_tokens_resolve="false" + done + if [[ "$all_tokens_resolve" == "true" ]]; then + manifest=$(printf '%s\n' $files_arg) + fi + fi + while IFS= read -r token; do if [[ "$token" == @* ]]; then path="${token#@}" - if [[ -f "$path" ]]; then - file_content+="=== ${path} ===${nl}$(cat "$path")${nl}${nl}" - elif [[ -d "$path" ]]; then - while IFS= read -r f; do - file_content+="=== ${f} ===${nl}$(cat "$f")${nl}${nl}" - done < <(find "$path" -type f \( \ - -name "*.ts" -o -name "*.js" -o -name "*.sh" -o -name "*.py" \ - -o -name "*.go" -o -name "*.rs" -o -name "*.md" -o -name "*.json" \ - \) 2>/dev/null | sort) + case "$path" in + ""|/*|../*|*/../*|*/..) echo "CONTEXT_OMITTED: path escapes project scope: ${path}" >&2; continue ;; + esac + path="${path#./}" + if [[ -L "$project_root/$path" ]]; then + echo "CONTEXT_OMITTED: symlink input is not followed: ${path}" >&2 + continue + fi + + local matched="false" + while IFS= read -r -d '' f; do + matched="true" + [[ -L "$project_root/$f" ]] && { echo "CONTEXT_OMITTED: symlink input is not followed: ${f}" >&2; continue; } + case "/$f" in + */.devsquad/*|*/.env|*/.env.*|*/credentials.json|*.pem|*.key) + echo "CONTEXT_OMITTED: sensitive or runtime path excluded: ${f}" >&2; continue ;; + esac + if git -C "$project_root" check-ignore --no-index -q -- "$f" 2>/dev/null; then + echo "CONTEXT_OMITTED: ignored path excluded: ${f}" >&2 + continue + fi + if ! grep -Iq . "$project_root/$f" 2>/dev/null && [[ -s "$project_root/$f" ]]; then + echo "CONTEXT_OMITTED: binary file excluded: ${f}" >&2 + continue + fi + file_bytes=$(wc -c < "$project_root/$f" | tr -d ' ') + if [[ "$file_bytes" -gt "$max_file_bytes" ]]; then + echo "CONTEXT_OMITTED: file exceeds ${max_file_bytes} byte limit: ${f} (${file_bytes} bytes)" >&2 + continue + fi + header_bytes=$(( ${#f} + 10 )) + if [[ $(( used_bytes + file_bytes + header_bytes )) -gt "$max_bytes" ]]; then + echo "CONTEXT_OMITTED: total context exceeds ${max_bytes} byte limit before: ${f}" >&2 + continue + fi + file_content+="=== ${f} ===${nl}$(cat "$project_root/$f")${nl}${nl}" + used_bytes=$(( used_bytes + file_bytes + header_bytes )) + done < <(git -C "$project_root" ls-files -z -- "$path" 2>/dev/null) + if [[ "$matched" == "false" ]]; then + echo "CONTEXT_OMITTED: no tracked files in scope: ${path}" >&2 fi fi - done + done <<< "$manifest" local final_prompt final_prompt=$(_gemini_final_prompt "$prompt" "$word_limit") diff --git a/plugin/lib/model-catalog.sh b/plugin/lib/model-catalog.sh index ee45235..f9550f3 100644 --- a/plugin/lib/model-catalog.sh +++ b/plugin/lib/model-catalog.sh @@ -61,15 +61,20 @@ refresh_model_catalog() { if [[ -n "$k_models" ]]; then k_status="ok"; else k_status="error"; fi fi + local old_json='{}' + [[ -f "$CATALOG_FILE" ]] && old_json=$(cat "$CATALOG_FILE" 2>/dev/null || echo '{}') local new_json new_json=$(jq -n \ --arg ts "$ts" \ --arg gs "$g_status" --arg g "$g_models" \ --arg ks "$k_status" --arg k "$k_models" \ + --argjson old "$old_json" \ '{ fetched_at: $ts, - gemini: { status: $gs, models: ($g | split("\n") | map(select(length > 0))) }, - grok: { status: $ks, models: ($k | split("\n") | map(select(length > 0))) }, + gemini: (if $gs == "ok" then { status: "ok", models: ($g | split("\n") | map(select(length > 0))), last_good_at: $ts, last_refresh_error: null } + else (($old.gemini // {models: []}) + {status: ($old.gemini.status // "unavailable"), last_refresh_error: $gs}) end), + grok: (if $ks == "ok" then { status: "ok", models: ($k | split("\n") | map(select(length > 0))), last_good_at: $ts, last_refresh_error: null } + else (($old.grok // {models: []}) + {status: ($old.grok.status // "unavailable"), last_refresh_error: $ks}) end), codex: { status: "unlistable", models: [] } }') @@ -104,11 +109,9 @@ refresh_model_catalog() { } # Map a tier to the best available model for a CLI, from the cached catalog. -# tier:fast -> cheap/fast family (flash|fast|mini|lite|haiku), -# highest version, prefer (Medium) then (Low) -# tier:frontier -> non-fast family matching pro|opus|max|ultra (fallback: -# any non-fast, then anything), highest version, prefer -# (High) then Thinking +# Structured entries declare family and compatible tiers. Legacy string entries +# are constrained to the harness family before applying their old local ranking; +# version numbers are never compared across model families. # Echoes "" when unresolvable — callers fall back to the CLI default. resolve_model_tier() { local cli="$1" tier="$2" @@ -116,19 +119,27 @@ resolve_model_tier() { command -v jq &>/dev/null || { echo ""; return 0; } local models - models=$(jq -r --arg c "$cli" '.[$c].models // [] | .[]' "$CATALOG_FILE" 2>/dev/null || true) + models=$(jq -r --arg c "$cli" --arg t "$tier" ' + .[$c].models // [] | .[] | + if type == "object" then + select((.family == $c) and ((.compatibility.tiers // []) | index($t))) | .id + else + select((ascii_downcase | startswith($c + "-") or startswith($c + " "))) | . + end' "$CATALOG_FILE" 2>/dev/null || true) [[ -n "$models" ]] || { echo ""; return 0; } local pool - if [[ "$tier" == "fast" ]]; then - pool=$(printf '%s\n' "$models" | grep -iE 'flash|fast|mini|lite|haiku' || true) - [[ -n "$pool" ]] || pool="$models" + if jq -e --arg c "$cli" '.[$c].models // [] | any(type == "object")' "$CATALOG_FILE" >/dev/null 2>&1; then + pool="$models" + elif [[ "$tier" == "fast" ]]; then + pool=$(printf '%s\n' "$models" | grep -iE '(^|[- (])(flash|fast|mini|lite|haiku)([- )]|$)' || true) + [[ -n "$pool" ]] || { echo ""; return 0; } else - pool=$(printf '%s\n' "$models" | grep -ivE 'flash|fast|mini|lite|haiku' | grep -iE 'pro|opus|max|ultra' || true) + pool=$(printf '%s\n' "$models" | grep -ivE '(^|[- (])(flash|fast|mini|lite|haiku)([- )]|$)' | grep -iE '(^|[- (])(pro|opus|max|ultra)([- )]|$)' || true) if [[ -z "$pool" ]]; then - pool=$(printf '%s\n' "$models" | grep -ivE 'flash|fast|mini|lite|haiku' || true) + pool=$(printf '%s\n' "$models" | grep -ivE '(^|[- (])(flash|fast|mini|lite|haiku)([- )]|$)' || true) fi - [[ -n "$pool" ]] || pool="$models" + [[ -n "$pool" ]] || { echo ""; return 0; } fi printf '%s\n' "$pool" | awk -v tier="$tier" ' diff --git a/test/core/test_m1.py b/test/core/test_m1.py new file mode 100644 index 0000000..2fe7341 --- /dev/null +++ b/test/core/test_m1.py @@ -0,0 +1,142 @@ +from __future__ import annotations + +import json +import os +import stat +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +import sys +sys.path.insert(0, str(CORE / "src")) + +from devsquad.adapters import AdapterManifest, classify_cli, prepare_cli +from devsquad.catalog import update_last_good +from devsquad.codex_protocol import NativeTurnState, model_list_request, parse_model_page +from devsquad.contracts import ContractError, validate_launch_payload +from devsquad.validation import validate_task + + +class M1ContractsTest(unittest.TestCase): + def manifest(self, name: str) -> AdapterManifest: + return AdapterManifest.load(CORE / "adapters" / name / "adapter.json") + + def fake_path(self, name: str) -> tuple[tempfile.TemporaryDirectory, Path]: + temp = tempfile.TemporaryDirectory() + path = Path(temp.name) / name + path.write_text("#!/bin/sh\nexit 0\n") + path.chmod(path.stat().st_mode | stat.S_IXUSR) + return temp, path + + def test_prepare_preserves_spaces_and_tsx_prompt(self): + temp, binary = self.fake_path("agy") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("antigravity"), prompt="Review ui/My Card.tsx", cwd=temp.name, model="Gemini Test", effort=None, permission="read_only", timeout_seconds=9) + self.assertIn("Review ui/My Card.tsx", spec.argv) + self.assertIn("Gemini Test", spec.argv) + self.assertEqual(spec.environment, {"DEVSQUAD_WORKER": "1"}) + + def test_explicit_unsupported_effort_fails_before_launch(self): + temp, binary = self.fake_path("grok") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + with self.assertRaisesRegex(ContractError, "unsupported or unverified effort"): + prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort="ultra", permission="read_only", timeout_seconds=9) + + def test_classifier_does_not_accept_empty_denied_or_malformed_exit_zero(self): + temp, binary = self.fake_path("grok") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + self.assertEqual(classify_cli(spec, returncode=0, stdout="", stderr="").execution_status, "malformed") + self.assertEqual(classify_cli(spec, returncode=0, stdout='{"type":"result","is_error":true,"error":"tool denied"}', stderr="").execution_status, "denied") + self.assertEqual(classify_cli(spec, returncode=0, stdout="not json", stderr="").execution_status, "malformed") + + def test_auth_precedes_rate_and_acceptance_is_separate(self): + temp, binary = self.fake_path("grok") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + result = classify_cli(spec, returncode=0, stdout='{"type":"result","is_error":true,"error":"401 quota rate limit"}', stderr="") + self.assertEqual(result.error_code, "AUTH_ERROR") + self.assertEqual(result.acceptance_status, "not_evaluated") + self.assertEqual(result.artifact_status, "unknown") + + def test_valid_auth_topic_is_deliverable_and_startup_only_is_not(self): + temp, binary = self.fake_path("grok") + self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + valid = '{"type":"result","result":"The author explains authentication.","is_error":false}' + self.assertEqual(classify_cli(spec, returncode=0, stdout=valid, stderr="").execution_status, "succeeded") + startup = '{"type":"system","subtype":"init"}' + self.assertEqual(classify_cli(spec, returncode=0, stdout=startup, stderr="").execution_status, "malformed") + + def test_codex_real_jsonl_shape_requires_agent_message(self): + temp, binary = self.fake_path("codex"); self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("codex"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + valid = '\n'.join([json.dumps({"type":"thread.started","thread_id":"x"}), json.dumps({"type":"item.completed","item":{"type":"agent_message","text":"done"}}), json.dumps({"type":"turn.completed","usage":{"input_tokens":1}})]) + self.assertEqual(classify_cli(spec, returncode=0, stdout=valid, stderr="").execution_status, "succeeded") + startup = json.dumps({"type":"thread.started","thread_id":"x"}) + "\n" + json.dumps({"type":"turn.completed","usage":{}}) + self.assertEqual(classify_cli(spec, returncode=0, stdout=startup, stderr="").execution_status, "malformed") + + def test_launch_round_trip_is_strict(self): + temp, binary = self.fake_path("grok"); self.addCleanup(temp.cleanup) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_cli(self.manifest("grok"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) + validate_launch_payload(spec.to_dict()) + invalid = spec.to_dict(); invalid["surprise"] = True + with self.assertRaises(ContractError): validate_launch_payload(invalid) + + def test_task_examples_validate_and_unknown_fields_fail(self): + root = Path(__file__).resolve().parents[2] + for name in ("branch-review.json", "issue-delivery.json"): + task = json.loads((root / "docs/plans/engineering-team/examples" / name).read_text()) + validate_task(task) + task["unknown"] = 1 + with self.assertRaises(ContractError): validate_task(task) + + +class NativeProtocolTest(unittest.TestCase): + def test_paginated_model_request_and_response(self): + self.assertEqual(model_list_request(2, "next")["params"]["cursor"], "next") + models, cursor = parse_model_page({"result": {"data": [{"id": "gpt-x"}], "nextCursor": "c2"}}) + self.assertEqual(models[0]["id"], "gpt-x") + self.assertEqual(cursor, "c2") + + def test_start_and_interrupt_ack_are_not_terminal(self): + state = NativeTurnState() + state.consume({"method": "turn/started", "params": {"turn": {"id": "t1"}}}) + state.consume({"method": "turn/interrupt/completed", "params": {}}) + self.assertFalse(state.terminal) + self.assertTrue(state.interrupted_acknowledged) + state.consume({"method": "turn/interrupted", "params": {}}) + self.assertTrue(state.terminal) + + def test_disconnect_or_malformed_page_is_visible(self): + with self.assertRaises(ContractError): + parse_model_page({"result": {"nextCursor": "never"}}) + + +class CatalogTest(unittest.TestCase): + def test_incomplete_refresh_retains_last_good(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "catalog.json" + first = update_last_good(path, harness="codex", version="1", models=[{"id": "gpt-a", "supportedReasoningEfforts": ["low"]}], complete=True) + retained = update_last_good(path, harness="codex", version="2", models=None, complete=False, error="timeout") + self.assertEqual(retained["models"], first["models"]) + self.assertEqual(retained["last_refresh"]["status"], "error") + + def test_discovered_model_is_unqualified_and_unknown_family_stays_unknown(self): + with tempfile.TemporaryDirectory() as tmp: + value = update_last_good(Path(tmp) / "catalog.json", harness="codex", version="1", models=[{"id": "surprise-9"}], complete=True) + self.assertEqual(value["models"][0]["qualification"], "unqualified") + self.assertIsNone(value["models"][0]["family"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/test_m1_legacy.sh b/test/test_m1_legacy.sh new file mode 100755 index 0000000..4165cd5 --- /dev/null +++ b/test/test_m1_legacy.sh @@ -0,0 +1,84 @@ +#!/usr/bin/env bash +set -u + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +PASS=0 FAIL=0 +ok() { PASS=$((PASS + 1)); } +bad() { FAIL=$((FAIL + 1)); echo " FAIL: $1"; } + +T=$(mktemp -d) +trap 'rm -rf "$T"' EXIT +mkdir -p "$T/bin" "$T/project/.devsquad" +cat > "$T/bin/codex" <<'EOF' +#!/usr/bin/env bash +case "${FAKE_MODE:-fast}" in + fast) printf 'ok' ;; + tree) + sh -c 'trap "" TERM; echo $$ > "$DESC_PID_FILE"; while :; do sleep 1; done' & + wait + ;; +esac +EOF +chmod +x "$T/bin/codex" + +# Force the portable path even on systems with timeout/gtimeout installed. +start=$(date +%s) +PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=fast \ + bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; invoke_codex hello 10 2' _ "$ROOT" > "$T/out" 2> "$T/err" +fast_elapsed=$(( $(date +%s) - start )) +[[ "$(cat "$T/out")" == "ok" ]] && ok || bad "portable watchdog output" +[[ "$fast_elapsed" -lt 2 ]] && ok || bad "portable fast call waited ${fast_elapsed}s for watchdog" + +# Model lookup remains optional when jq is absent from PATH. +mkdir -p "$T/nojq" +ln -s /bin/bash "$T/nojq/bash" +ln -s /usr/bin/dirname "$T/nojq/dirname" +PATH="$T/nojq" CLAUDE_PROJECT_DIR="$T/project" /bin/bash -c \ + 'source "$1/plugin/lib/codex-wrapper.sh"; [[ -z "$(_resolve_codex_model)" ]]' _ "$ROOT" && ok || bad "jq-absent model fallback" + +DESC_PID_FILE="$T/desc.pid"; export DESC_PID_FILE +start=$(date +%s) +PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=tree \ + bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; invoke_codex hello 10 1' _ "$ROOT" > "$T/tree.out" 2> "$T/tree.err" || rc=$? +elapsed=$(( $(date +%s) - start )) +[[ "${rc:-0}" -eq 1 ]] && grep -q '^TIMEOUT:' "$T/tree.err" && ok || bad "portable timeout classification" +[[ "$elapsed" -lt 4 ]] && ok || bad "portable timeout elapsed ${elapsed}s" +if [[ -s "$DESC_PID_FILE" ]] && kill -0 "$(cat "$DESC_PID_FILE")" 2>/dev/null; then + bad "portable timeout left descendant alive" +else + ok +fi + +# Newline manifests preserve spaces and enumerate TSX. +mkdir -p "$T/project/ui/My Folder" +printf 'export const Card = 1;\n' > "$T/project/ui/My Folder/Card.tsx" +git -C "$T/project" init -q +git -C "$T/project" add "ui/My Folder/Card.tsx" +cat > "$T/bin/agy" <<'EOF' +#!/usr/bin/env bash +cat +printf '{"ok":true}\n' +EOF +chmod +x "$T/bin/agy" +PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" \ + bash -c 'cd "$1"; source "$2/plugin/lib/gemini-wrapper.sh"; invoke_gemini_with_files "@ui/My Folder" inspect 10 2' _ "$T/project" "$ROOT" > "$T/context" 2>/dev/null +grep -q 'Card.tsx' "$T/context" && ok || bad "TSX path with spaces omitted" + +# Ignored, oversized, escaping and symlink inputs are reported rather than +# silently read or truncated. +printf 'ignored.txt\n' > "$T/project/.gitignore" +printf 'secret\n' > "$T/project/ignored.txt" +printf '123456789\n' > "$T/project/large.ts" +ln -s /etc/passwd "$T/project/escape.ts" +git -C "$T/project" add .gitignore large.ts +git -C "$T/project" add -f ignored.txt escape.ts +PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" DEVSQUAD_CONTEXT_MAX_FILE_BYTES=4 \ + bash -c 'source "$1/plugin/lib/gemini-wrapper.sh"; invoke_gemini_with_files "$2" inspect 10 2' _ "$ROOT" \ + $'@ignored.txt\n@large.ts\n@escape.ts\n@../outside' > "$T/omitted.out" 2> "$T/omitted.err" +grep -q 'ignored path excluded: ignored.txt' "$T/omitted.err" && ok || bad "ignored file omission not reported" +grep -q 'file exceeds 4 byte limit: large.ts' "$T/omitted.err" && ok || bad "byte limit omission not reported" +grep -q 'symlink input is not followed: escape.ts' "$T/omitted.err" && ok || bad "symlink omission not reported" +grep -q 'path escapes project scope: ../outside' "$T/omitted.err" && ok || bad "path escape not reported" + +echo " m1_legacy: ${PASS} passed, ${FAIL} failed" +[[ "$FAIL" -eq 0 ]] diff --git a/test/test_models.sh b/test/test_models.sh index d226e5a..5f43b1b 100644 --- a/test/test_models.sh +++ b/test/test_models.sh @@ -72,10 +72,31 @@ CATEOF CAT="$PLUGIN_ROOT/lib/model-catalog.sh" assert_eq "tier fast picks newest flash" "$(bash "$CAT" resolve gemini fast)" "Gemini 4.0 Flash (Medium)" -assert_eq "tier frontier picks highest ver" "$(bash "$CAT" resolve gemini frontier)" "Claude Opus 4.6 (Thinking)" +assert_eq "tier frontier stays in family" "$(bash "$CAT" resolve gemini frontier)" "Gemini 3.1 Pro (High)" assert_eq "grok fast" "$(bash "$CAT" resolve grok fast)" "grok-composer-2.5-fast" assert_eq "grok frontier non-fast fallback" "$(bash "$CAT" resolve grok frontier)" "grok-build" +# Structured compatibility is authoritative: unrelated families and entries +# without a declared tier cannot participate in selection. +cat > "$CATDIR/models.json" <<'CATEOF' +{"fetched_at":"2026-07-06T00:00:00Z", + "gemini":{"status":"ok","models":[ + {"id":"gemini-pro","family":"gemini","compatibility":{"tiers":["frontier"]}}, + {"id":"claude-opus-99","family":"claude","compatibility":{"tiers":["frontier"]}}, + {"id":"gemini-unknown","family":"gemini","compatibility":{"tiers":[]}} + ]},"grok":{"status":"ok","models":[]},"codex":{"status":"unlistable","models":[]}} +CATEOF +assert_eq "structured compatibility stays in family" "$(bash "$CAT" resolve gemini frontier)" "gemini-pro" +assert_eq "structured unsupported tier is empty" "$(bash "$CAT" resolve gemini fast)" "" + +# Restore the legacy-string fixture for the adapter compatibility checks. +cat > "$CATDIR/models.json" <<'CATEOF' +{"fetched_at":"2026-07-06T00:00:00Z", + "gemini":{"status":"ok","models":["Gemini 3.5 Flash (Medium)","Gemini 4.0 Flash (Medium)","Gemini 3.1 Pro (High)","Claude Opus 4.6 (Thinking)"]}, + "grok":{"status":"ok","models":["grok-composer-2.5-fast","grok-build"]}, + "codex":{"status":"unlistable","models":[]}} +CATEOF + # Adapter integration: tier pin in agent_models resolves through the catalog T4=$(mktemp -d); mkdir -p "$T4/.devsquad" printf '%s' '{"agent_models":{"gemini-reader":"tier:fast"}}' > "$T4/.devsquad/config.json" @@ -102,5 +123,23 @@ else PASS=$((PASS + 1)) fi +# A failed discovery refresh records the error while retaining the prior +# successful model set; it is not interpreted as every model being removed. +REFRESH_BIN=$(mktemp -d) +cat > "$REFRESH_BIN/agy" <<'EOF' +#!/usr/bin/env bash +exit 1 +EOF +cat > "$REFRESH_BIN/grok" <<'EOF' +#!/usr/bin/env bash +exit 1 +EOF +chmod +x "$REFRESH_BIN/agy" "$REFRESH_BIN/grok" +before_models=$(jq -c '.gemini.models' "$CATDIR/models.json") +PATH="$REFRESH_BIN:$PATH" bash -c 'source "$1"; refresh_model_catalog' _ "$CAT" >/dev/null 2>&1 +after_models=$(jq -c '.gemini.models' "$CATDIR/models.json") +assert_eq "failed refresh retains last-good models" "$after_models" "$before_models" +assert_eq "failed refresh records error" "$(jq -r '.gemini.last_refresh_error' "$CATDIR/models.json")" "error" + echo " models: ${PASS} passed, ${FAIL} failed" [ "$FAIL" -eq 0 ] From 0826d13872ce7bf9229e8382b6dcff093165486e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 6 Sep 2026 18:33:32 +0530 Subject: [PATCH 008/197] docs: record M1 verification checkpoint --- .gitignore | 1 + docs/plans/engineering-team/M1-STATUS.md | 22 +++ docs/plans/engineering-team/backlog.json | 16 +- .../M1-invocation-core-2026-09-06.json | 43 +++++ plugin/core/build/lib/devsquad/__init__.py | 3 - plugin/core/build/lib/devsquad/adapters.py | 151 ------------------ plugin/core/build/lib/devsquad/catalog.py | 60 ------- plugin/core/build/lib/devsquad/cli.py | 85 ---------- .../core/build/lib/devsquad/codex_protocol.py | 76 --------- plugin/core/build/lib/devsquad/contracts.py | 103 ------------ plugin/core/build/lib/devsquad/validation.py | 53 ------ 11 files changed, 79 insertions(+), 534 deletions(-) create mode 100644 docs/plans/engineering-team/M1-STATUS.md create mode 100644 docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json delete mode 100644 plugin/core/build/lib/devsquad/__init__.py delete mode 100644 plugin/core/build/lib/devsquad/adapters.py delete mode 100644 plugin/core/build/lib/devsquad/catalog.py delete mode 100644 plugin/core/build/lib/devsquad/cli.py delete mode 100644 plugin/core/build/lib/devsquad/codex_protocol.py delete mode 100644 plugin/core/build/lib/devsquad/contracts.py delete mode 100644 plugin/core/build/lib/devsquad/validation.py diff --git a/.gitignore b/.gitignore index 95bb347..73f1de5 100644 --- a/.gitignore +++ b/.gitignore @@ -10,6 +10,7 @@ __pycache__/ *.py[cod] .pytest_cache/ *.egg-info/ +plugin/core/build/ # Logs *.log diff --git a/docs/plans/engineering-team/M1-STATUS.md b/docs/plans/engineering-team/M1-STATUS.md new file mode 100644 index 0000000..a486da7 --- /dev/null +++ b/docs/plans/engineering-team/M1-STATUS.md @@ -0,0 +1,22 @@ +# M1 implementation status + +M1 is implemented at checkpoint `572e452` and remains **in progress** pending +independent review of the complete gate. M2 process ownership has not started. + +| Requirement | Evidence | Status | +|---|---|---| +| Python 3.11+ package, launcher, strict v1 inputs | `plugin/core`, 13 standard-library unit tests, source and installed-wheel launch | verified | +| Requested and observed identity remain separate | `ExecutionIdentity`, `LaunchSpec`, `NormalizedResult` | verified offline | +| Execution, artifact and acceptance states remain separate | normalized-result schema and classifier tests | verified offline | +| Explicit model/effort/permission preparation | manifest builders reject unknown effort pairs; argv preserves spaces | verified offline | +| Codex native metadata and lifecycle dialect | installed 0.135.0 schema generation, live paginated `model/list`, fake event-state tests | verified | +| Catalog last-good retention and unqualified discovery | catalog unit and legacy shell tests | verified offline | +| Legacy wrapper API and four error prefixes | existing discovery suite, 38 wrapper assertions | verified offline | +| Portable watchdog prompt return and descendant cleanup | forced portable-path Bash 3.2 tests | verified offline | +| Tracked bounded Antigravity context | Git inventory tests cover spaces, TSX/JSX, ignored/binary/symlink/escape/size omissions | verified offline | +| Bounded real starting-profile smoke | Codex `gpt-5.5`, low effort, read-only, ephemeral JSONL invocation | verified live | +| Independent gate review | Root review is active; findings were folded into the checkpoint | pending final review | + +The Python bridge returns argv and parser policy only. It does not spawn a +worker or implement a watchdog. M2 remains the sole owner of process sessions, +timeouts, cancellation, draining and reaping for new runs. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index b134d22..62e71c0 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -10,16 +10,26 @@ "model_lifecycle_and_native_adapters": "MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md", "execution_brief": "SOL-HANDOFF.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], - "status": "planned", + "status": "in_progress", "next_milestone": "M1", "milestones": [ { "id": "M1", "title": "Truthful reusable adapter invocation", "depends_on": [], - "status": "pending", + "status": "in_progress", "acceptance_section": "M1 — Make invocation truthful and reusable", - "evidence": [], + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "572e452", + "command_or_action": "offline core/legacy suites, temporary wheel install, live Codex app-server metadata and read-only invocation probe", + "outcome": "M1 implementation and stated probes pass; independent gate review remains open", + "artifact": "evidence/M1-invocation-core-2026-09-06.json", + "recorded_at": "2026-09-06T13:02:37Z", + "availability": "portable_redacted" + } + ], "blocker": null }, { diff --git a/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json b/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json new file mode 100644 index 0000000..fc16268 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json @@ -0,0 +1,43 @@ +{ + "schema_version": 1, + "milestone": "M1", + "status": "in_progress", + "implementation_revision": "572e452", + "implementation_tree": "26abe92ee4c89dbd481ada20ebd02fb9d6898f51", + "recorded_at": "2026-09-06T13:02:37Z", + "evidence": [ + { + "kind": "offline_test", + "command_or_action": "/bin/bash test/run.sh", + "outcome": "10 test files passed; 192 shell assertions passed including Bash 3.2 portable timeout and descendant cleanup", + "availability": "tracked tests" + }, + { + "kind": "offline_test", + "command_or_action": "PYTHONDONTWRITEBYTECODE=1 python3 -m unittest discover -s test/core -v", + "outcome": "13 tests passed", + "availability": "tracked tests" + }, + { + "kind": "package_test", + "command_or_action": "bundled Python 3.12: pip wheel --no-build-isolation --no-deps plugin/core; install into temporary venv; squad --version; squad doctor --json", + "outcome": "wheel built and installed; packaged adapter resources resolved outside checkout; squad 0.1.0 reported installed harnesses", + "availability": "reproducible locally" + }, + { + "kind": "live_metadata", + "command_or_action": "Codex 0.135.0 app-server initialize then model/list with limit 100", + "outcome": "initialized; complete single page with 5 account-visible models and model-scoped reasoning efforts; default gpt-5.5/medium", + "availability": "redacted portable summary" + }, + { + "kind": "live_smoke", + "command_or_action": "codex exec --json --sandbox read-only --ephemeral --model gpt-5.5 with model_reasoning_effort=low and a no-tools response probe", + "outcome": "exit 0; thread.started, turn.started, item.completed and turn.completed observed; final message DEVSQUAD_M1_PROBE_OK", + "availability": "redacted portable summary" + } + ], + "residual_blockers": [ + "Independent M1 gate review is not yet complete." + ] +} diff --git a/plugin/core/build/lib/devsquad/__init__.py b/plugin/core/build/lib/devsquad/__init__.py deleted file mode 100644 index 22a4c4a..0000000 --- a/plugin/core/build/lib/devsquad/__init__.py +++ /dev/null @@ -1,3 +0,0 @@ -"""DevSquad's surface-independent local core.""" - -__version__ = "0.1.0" diff --git a/plugin/core/build/lib/devsquad/adapters.py b/plugin/core/build/lib/devsquad/adapters.py deleted file mode 100644 index 152c27e..0000000 --- a/plugin/core/build/lib/devsquad/adapters.py +++ /dev/null @@ -1,151 +0,0 @@ -"""Manifest-driven M1 adapter preparation and output classification.""" - -from __future__ import annotations - -import json -import os -import re -import shutil -import subprocess -from dataclasses import dataclass -from pathlib import Path -from typing import Any - -from .contracts import ContractError, ExecutionIdentity, LaunchSpec, NormalizedResult, ProfileUnsupported, SCHEMA_VERSION - -ERROR_PATTERNS = ( - ("AUTH_ERROR", re.compile(r"auth|unauthorized|ineligible|\b401\b|\b403\b", re.I)), - ("RATE_LIMITED", re.compile(r"rate.?limit|quota|resource.?exhausted|too many requests|\b429\b", re.I)), -) - - -@dataclass(frozen=True) -class AdapterManifest: - name: str - transport: str - binary_candidates: tuple[str, ...] - model_provider: str | None - efforts_by_model: dict[str, tuple[str, ...]] - permission_profiles: dict[str, tuple[str, ...]] - output_format: str - - @classmethod - def load(cls, path: Path) -> "AdapterManifest": - raw = json.loads(path.read_text()) - required = {"schema_version", "name", "transport", "binary_candidates", "capabilities"} - missing = required - raw.keys() - if missing or raw["schema_version"] != 1: - raise ContractError(f"invalid adapter manifest: missing={sorted(missing)}") - capabilities = raw["capabilities"] - return cls( - name=raw["name"], transport=raw["transport"], - binary_candidates=tuple(raw["binary_candidates"]), - model_provider=raw.get("model_provider"), - efforts_by_model={k: tuple(v) for k, v in capabilities.get("efforts_by_model", {}).items()}, - permission_profiles={k: tuple(v) for k, v in raw.get("permission_profiles", {}).items()}, - output_format=raw.get("output_format", "text"), - ) - - def resolve_binary(self) -> str | None: - return next((p for name in self.binary_candidates if (p := shutil.which(name))), None) - - -def _permission_args(manifest: AdapterManifest, permission: str) -> tuple[str, ...]: - try: - return manifest.permission_profiles[permission] - except KeyError as exc: - raise ContractError(f"unsupported permission profile: {permission}") from exc - - -def prepare_cli( - manifest: AdapterManifest, *, prompt: str, cwd: str, model: str | None, - effort: str | None, permission: str, timeout_seconds: int, stdin_path: str | None = None, -) -> LaunchSpec: - binary = manifest.resolve_binary() - if not binary: - raise ContractError(f"adapter unavailable: {manifest.name}") - if effort is not None: - supported = manifest.efforts_by_model.get(model or "") - if supported is None or effort not in supported: - raise ProfileUnsupported(f"unsupported or unverified effort {effort!r} for {manifest.name} model {model!r}") - args: list[str] - if manifest.name == "codex": - sandbox = "read-only" if permission == "read_only" else "workspace-write" - args = [binary, "exec", "--json", "--cd", cwd, "--sandbox", sandbox] - if model: - args += ["--model", model] - if effort: - args += ["-c", f'model_reasoning_effort="{effort}"'] - args += [prompt] - elif manifest.name == "antigravity": - args = [binary, "--print", prompt, "--output-format", "json", "--mode", "plan" if permission == "read_only" else "accept-edits"] - if model: - args += ["--model", model] - if effort: - args += ["--effort", effort] - elif manifest.name == "grok": - args = [binary, "--single", prompt, "--output-format", "json", "--cwd", cwd, "--permission-mode", "plan" if permission == "read_only" else "acceptEdits", "--no-subagents"] - if model: - args += ["--model", model] - if effort: - args += ["--reasoning-effort", effort] - else: - raise ContractError(f"no argv builder for adapter: {manifest.name}") - args.extend(_permission_args(manifest, permission)) - requested = ExecutionIdentity( - harness=manifest.name, harness_version=None, model_provider=manifest.model_provider, - model_family=None, model=model, effort=effort, permissions=permission, - verification="unverified", - ) - return LaunchSpec(SCHEMA_VERSION, manifest.name, "cli_exec", tuple(args), str(Path(cwd).resolve()), stdin_path, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) - - -def _provider_records(stdout: str) -> tuple[list[dict[str, Any]], bool, str | None]: - records: list[dict[str, Any]] = [] - deliverable = False - error_text = None - for line in stdout.splitlines(): - if not line.strip(): - continue - item = json.loads(line) - if not isinstance(item, dict): - raise json.JSONDecodeError("record is not an object", line, 0) - records.append(item) - if item.get("is_error") is True or item.get("error"): - error_text = str(item.get("error") or item.get("result") or item.get("message")) - kind = item.get("type") - if kind in {"result", "assistant_message", "turn.completed"} and item.get("is_error") is not True: - payload = item.get("result") or item.get("message") or item.get("text") - deliverable = isinstance(payload, str) and bool(payload.strip()) - return records, deliverable, error_text - - -def classify_cli(spec: LaunchSpec, *, returncode: int, stdout: str, stderr: str, timed_out: bool = False) -> NormalizedResult: - code = next((code for code, pattern in ERROR_PATTERNS if pattern.search(stderr)), None) - status = "succeeded" - if code == "AUTH_ERROR" or code == "RATE_LIMITED": - status = "failed" - elif timed_out or returncode in (124, 137, 143): - status, code = "timed_out", "TIMEOUT" - elif returncode != 0: - status, code = "failed", "CLI_ERROR" - elif not stdout.strip(): - status, code = "malformed", "CLI_ERROR" - elif spec.adapter in {"codex", "antigravity", "grok"}: - try: - _, deliverable, provider_error = _provider_records(stdout) - if provider_error: - code = next((candidate for candidate, pattern in ERROR_PATTERNS if pattern.search(provider_error)), "CLI_ERROR") - status = "denied" if re.search(r"permission denied|tool (?:use )?denied|not allowed", provider_error, re.I) else "failed" - elif not deliverable: - status, code = "malformed", "CLI_ERROR" - except json.JSONDecodeError: - status, code = "malformed", "CLI_ERROR" - return NormalizedResult(SCHEMA_VERSION, status, code, stdout if stdout else None, "unknown", "not_evaluated", spec.requested, None) - - -def harness_version(binary: str) -> str | None: - try: - return subprocess.run([binary, "--version"], text=True, capture_output=True, timeout=3, check=False).stdout.strip() or None - except (OSError, subprocess.TimeoutExpired): - return None diff --git a/plugin/core/build/lib/devsquad/catalog.py b/plugin/core/build/lib/devsquad/catalog.py deleted file mode 100644 index 59bc94c..0000000 --- a/plugin/core/build/lib/devsquad/catalog.py +++ /dev/null @@ -1,60 +0,0 @@ -"""Structured, last-good model catalog handling for M1.""" - -from __future__ import annotations - -import hashlib -import json -import os -from datetime import datetime, timezone -from pathlib import Path -from typing import Any, Iterable - -from .contracts import ContractError - - -def model_fingerprint(harness: str, version: str | None, model: dict[str, Any]) -> str: - stable = {"harness": harness, "version": version, "model": model} - return hashlib.sha256(json.dumps(stable, sort_keys=True, separators=(",", ":")).encode()).hexdigest() - - -def normalize_models(harness: str, version: str | None, models: Iterable[dict[str, Any]]) -> list[dict[str, Any]]: - normalized = [] - for raw in models: - model_id = raw.get("id") or raw.get("model") - if not isinstance(model_id, str) or not model_id: - raise ContractError("catalog model missing id") - effort_values = raw.get("supportedReasoningEfforts") or raw.get("supported_reasoning_efforts") or [] - efforts = [item.get("reasoningEffort") if isinstance(item, dict) else item for item in effort_values] - normalized.append({ - "id": model_id, - "display_name": raw.get("displayName") or raw.get("display_name") or model_id, - "family": raw.get("family"), - "default_effort": raw.get("defaultReasoningEffort") or raw.get("default_reasoning_effort"), - "supported_efforts": efforts, - "modalities": raw.get("inputModalities") or raw.get("input_modalities") or [], - "is_default": bool(raw.get("isDefault") or raw.get("is_default")), - "qualification": "unqualified", - "fingerprint": model_fingerprint(harness, version, raw), - }) - return normalized - - -def update_last_good(path: Path, *, harness: str, version: str | None, models: Iterable[dict[str, Any]] | None, complete: bool, error: str | None = None) -> dict[str, Any]: - old = json.loads(path.read_text()) if path.exists() else None - now = datetime.now(timezone.utc).isoformat() - if error or not complete or models is None: - if old: - old["last_refresh"] = {"at": now, "status": "error" if error else "incomplete", "error": error} - _atomic_json(path, old) - return old - raise ContractError(error or "catalog response incomplete and no last-good snapshot exists") - value = {"schema_version": 1, "harness": harness, "harness_version": version, "fetched_at": now, "complete": True, "models": normalize_models(harness, version, models), "last_refresh": {"at": now, "status": "ok", "error": None}} - _atomic_json(path, value) - return value - - -def _atomic_json(path: Path, value: dict[str, Any]) -> None: - path.parent.mkdir(parents=True, exist_ok=True) - tmp = path.with_name(f"{path.name}.tmp.{os.getpid()}") - tmp.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") - os.replace(tmp, path) diff --git a/plugin/core/build/lib/devsquad/cli.py b/plugin/core/build/lib/devsquad/cli.py deleted file mode 100644 index e1be5bc..0000000 --- a/plugin/core/build/lib/devsquad/cli.py +++ /dev/null @@ -1,85 +0,0 @@ -"""Small M1 command surface: version, doctor, prepare and classify.""" - -from __future__ import annotations - -import argparse -import json -import sys -from pathlib import Path - -from . import __version__ -from .adapters import AdapterManifest, classify_cli, harness_version, prepare_cli -from .contracts import ContractError, envelope, error_payload - -SOURCE_ROOT = Path(__file__).resolve().parents[2] -CORE_ROOT = SOURCE_ROOT if (SOURCE_ROOT / "adapters").is_dir() else Path(sys.prefix) / "share" / "devsquad" - - -def manifests() -> list[tuple[Path, AdapterManifest]]: - return [(path, AdapterManifest.load(path)) for path in sorted((CORE_ROOT / "adapters").glob("*/adapter.json"))] - - -def command_doctor(_: argparse.Namespace) -> tuple[dict, int]: - rows = [] - for path, manifest in manifests(): - binary = manifest.resolve_binary() - rows.append({"adapter": manifest.name, "transport": manifest.transport, "status": "unverified" if binary else "unavailable", "binary": binary, "version": harness_version(binary) if binary else None, "manifest": str(path)}) - ready = any(row["status"] != "unavailable" for row in rows) - return envelope(data={"core_version": __version__, "ready": ready, "adapters": rows}), 0 if ready else 1 - - -def command_prepare(args: argparse.Namespace) -> dict: - manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") - spec = prepare_cli(manifest, prompt=args.prompt, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) - return envelope(data=spec.to_dict()), 0 - - -def command_classify(args: argparse.Namespace) -> dict: - manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") - spec = prepare_cli(manifest, prompt="classification", cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) - result = classify_cli(spec, returncode=args.returncode, stdout=Path(args.stdout_file).read_text(), stderr=Path(args.stderr_file).read_text()) - return envelope(data=result.to_dict()), 0 - - -class ContractParser(argparse.ArgumentParser): - def error(self, message: str) -> None: - raise ContractError(message) - - -def parser() -> argparse.ArgumentParser: - p = ContractParser(prog="squad") - p.add_argument("--version", action="version", version=f"squad {__version__}") - sub = p.add_subparsers(dest="command", required=True) - doctor = sub.add_parser("doctor"); doctor.add_argument("--json", action="store_true"); doctor.set_defaults(func=command_doctor) - for name, fn in (("prepare", command_prepare), ("classify", command_classify)): - cmd = sub.add_parser(name) - cmd.add_argument("adapter", choices=("codex", "antigravity", "grok")) - cmd.add_argument("--cwd", default=str(Path.cwd())) - cmd.add_argument("--model") - cmd.add_argument("--effort") - cmd.add_argument("--permission", choices=("read_only", "workspace_write"), default="read_only") - cmd.add_argument("--timeout", type=int, default=90) - if name == "prepare": - cmd.add_argument("--prompt", required=True) - else: - cmd.add_argument("--returncode", type=int, required=True) - cmd.add_argument("--stdout-file", required=True) - cmd.add_argument("--stderr-file", required=True) - cmd.set_defaults(func=fn) - return p - - -def main(argv: list[str] | None = None) -> int: - try: - args = parser().parse_args(argv) - response, code = args.func(args) - print(json.dumps(response, sort_keys=True)) - return code - except (ContractError, OSError, json.JSONDecodeError) as exc: - code = getattr(exc, "code", "INPUT_INVALID") - print(json.dumps(envelope(error=error_payload(code, str(exc))), sort_keys=True)) - return 64 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/plugin/core/build/lib/devsquad/codex_protocol.py b/plugin/core/build/lib/devsquad/codex_protocol.py deleted file mode 100644 index e7eedd2..0000000 --- a/plugin/core/build/lib/devsquad/codex_protocol.py +++ /dev/null @@ -1,76 +0,0 @@ -"""Typed JSON-RPC preparation/state normalization for Codex app-server. - -This module does not spawn or supervise the server. M2 gives it a connected -stdio stream owned by the run supervisor. -""" - -from __future__ import annotations - -from dataclasses import dataclass, field -from typing import Any - -from .contracts import ContractError - - -def request(request_id: int, method: str, params: dict[str, Any] | None = None) -> dict[str, Any]: - return {"jsonrpc": "2.0", "id": request_id, "method": method, "params": params or {}} - - -def initialize_request(request_id: int = 1) -> dict[str, Any]: - return request(request_id, "initialize", {"clientInfo": {"name": "devsquad", "version": "0.1.0"}, "capabilities": {"experimentalApi": True}}) - - -def model_list_request(request_id: int, cursor: str | None = None, limit: int = 100) -> dict[str, Any]: - params: dict[str, Any] = {"limit": limit} - if cursor: - params["cursor"] = cursor - return request(request_id, "model/list", params) - - -def turn_start_request(request_id: int, *, thread_id: str, prompt: str, model: str | None, effort: str | None, cwd: str, sandbox: str) -> dict[str, Any]: - params: dict[str, Any] = {"threadId": thread_id, "input": [{"type": "text", "text": prompt}], "cwd": cwd, "permissions": sandbox} - if model: - params["model"] = model - if effort: - params["effort"] = effort - return request(request_id, "turn/start", params) - - -@dataclass -class NativeTurnState: - thread_id: str | None = None - turn_id: str | None = None - terminal: bool = False - interrupted_acknowledged: bool = False - output: list[str] = field(default_factory=list) - events: list[dict[str, Any]] = field(default_factory=list) - - def consume(self, message: dict[str, Any]) -> None: - if not isinstance(message, dict): - raise ContractError("native message must be an object") - self.events.append(message) - method = message.get("method", "") - params = message.get("params") or message.get("result") or {} - if method in {"thread/started", "thread/start/completed"}: - self.thread_id = params.get("thread", {}).get("id") or params.get("threadId") or self.thread_id - if method in {"turn/started", "turn/start/completed"}: - self.turn_id = params.get("turn", {}).get("id") or params.get("turnId") or self.turn_id - if method in {"item/agentMessage/delta", "turn/output/delta"}: - self.output.append(params.get("delta", "")) - if method in {"turn/interrupt/completed", "turn/interrupted/acknowledged"}: - self.interrupted_acknowledged = True - if method in {"turn/completed", "review/completed", "turn/failed", "turn/interrupted"}: - self.terminal = True - - -def parse_model_page(response: dict[str, Any]) -> tuple[list[dict[str, Any]], str | None]: - if "error" in response: - raise ContractError(f"model/list failed: {response['error']}") - result = response.get("result") - if not isinstance(result, dict): - raise ContractError("model/list missing result") - models = result.get("data") or result.get("models") - if not isinstance(models, list): - raise ContractError("model/list is incomplete") - cursor = result.get("nextCursor") or result.get("next_cursor") - return models, cursor diff --git a/plugin/core/build/lib/devsquad/contracts.py b/plugin/core/build/lib/devsquad/contracts.py deleted file mode 100644 index 9fca7dc..0000000 --- a/plugin/core/build/lib/devsquad/contracts.py +++ /dev/null @@ -1,103 +0,0 @@ -"""Strict M1 contracts for prepared invocations and normalized results.""" - -from __future__ import annotations - -from dataclasses import asdict, dataclass, field -from pathlib import Path -from typing import Any, Literal - -SCHEMA_VERSION = 1 -Transport = Literal["cli_exec", "native_protocol"] -Verification = Literal["verified", "unverified", "unavailable", "unknown"] - - -class ContractError(ValueError): - """Raised before launch when an invocation contract is unsupported.""" - - code = "INPUT_INVALID" - - -class ProfileUnsupported(ContractError): - code = "PROFILE_UNSUPPORTED" - - -@dataclass(frozen=True) -class ExecutionIdentity: - harness: str - harness_version: str | None - model_provider: str | None - model_family: str | None - model: str | None - effort: str | None - tools: tuple[str, ...] = () - permissions: str = "read_only" - account_pool: str | None = None - verification: Verification = "unknown" - - -@dataclass(frozen=True) -class LaunchSpec: - """A launch description. M2 owns spawning, timeout, cancellation and reaping.""" - - schema_version: int - adapter: str - transport: Transport - argv: tuple[str, ...] - cwd: str - stdin_path: str | None - timeout_seconds: int - requested: ExecutionIdentity - environment: dict[str, str] = field(default_factory=dict) - - def __post_init__(self) -> None: - if self.schema_version != SCHEMA_VERSION: - raise ContractError("unsupported schema_version") - if self.transport not in ("cli_exec", "native_protocol"): - raise ContractError("unsupported transport") - if not self.argv or not all(isinstance(v, str) and v for v in self.argv): - raise ContractError("argv must be a non-empty string array") - if self.timeout_seconds <= 0: - raise ContractError("timeout_seconds must be positive") - if not Path(self.cwd).is_absolute(): - raise ContractError("cwd must be absolute") - - def to_dict(self) -> dict[str, Any]: - return asdict(self) - - -@dataclass(frozen=True) -class NormalizedResult: - """Execution evidence without conflating artifacts or acceptance.""" - - schema_version: int - execution_status: Literal[ - "succeeded", "failed", "timed_out", "interrupted", "denied", "malformed" - ] - error_code: str | None - output: str | None - artifact_status: Literal["present", "missing", "not_required", "unknown"] - acceptance_status: Literal["pending", "accepted", "rejected", "not_evaluated"] - requested: ExecutionIdentity - observed: ExecutionIdentity | None - native_ids: dict[str, str] = field(default_factory=dict) - events: tuple[dict[str, Any], ...] = () - - def to_dict(self) -> dict[str, Any]: - return asdict(self) - - -def envelope(*, data: Any = None, error: dict[str, Any] | None = None) -> dict[str, Any]: - if (data is None) == (error is None): - raise ContractError("exactly one of data and error is required") - return {"schema_version": SCHEMA_VERSION, "ok": error is None, "data": data, "error": error} - - -def error_payload(code: str, message: str, *, retryable: bool = False, details: dict[str, Any] | None = None) -> dict[str, Any]: - return {"code": code, "message": message, "retryable": retryable, "details": details or {}} - - -def validate_launch_payload(value: dict[str, Any]) -> None: - expected = {"schema_version", "adapter", "transport", "argv", "cwd", "stdin_path", "timeout_seconds", "requested", "environment"} - if set(value) != expected: - raise ContractError(f"LaunchSpec fields differ: {sorted(set(value) ^ expected)}") - LaunchSpec(**{**value, "argv": tuple(value["argv"]), "requested": ExecutionIdentity(**{**value["requested"], "tools": tuple(value["requested"]["tools"])})}) diff --git a/plugin/core/build/lib/devsquad/validation.py b/plugin/core/build/lib/devsquad/validation.py deleted file mode 100644 index dd0c585..0000000 --- a/plugin/core/build/lib/devsquad/validation.py +++ /dev/null @@ -1,53 +0,0 @@ -"""Dependency-free strict validation for M1 public fixtures.""" - -from __future__ import annotations -from pathlib import Path -from typing import Any -from .contracts import ContractError - -TASK_FIELDS = {"schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin", "review"} - - -def _exact(value: dict[str, Any], allowed: set[str], required: set[str], label: str) -> None: - if not isinstance(value, dict): - raise ContractError(f"{label} must be an object") - unknown, missing = set(value) - allowed, required - set(value) - if unknown or missing: - raise ContractError(f"{label} fields invalid: unknown={sorted(unknown)} missing={sorted(missing)}") - - -def _relative(path: str, label: str) -> None: - p = Path(path) - if p.is_absolute() or ".." in p.parts: - raise ContractError(f"{label} must be repository-relative without traversal") - - -def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) -> None: - _exact(value, TASK_FIELDS, TASK_FIELDS - {"review"}, "task") - if value["schema_version"] != 1 or value["workflow"] not in {"branch-review", "issue-delivery"}: - raise ContractError("unsupported task schema or workflow") - project = value["project"] - _exact(project, {"repo_path", "base_ref", "target_ref"}, {"repo_path", "base_ref", "target_ref"}, "project") - if not Path(project["repo_path"]).is_absolute() or (require_existing_repo and not (Path(project["repo_path"]) / ".git").exists()): - raise ContractError("project.repo_path must be an existing absolute Git repository") - if not isinstance(value["goal"], str) or not value["goal"].strip() or not isinstance(value["task_class"], str) or not value["task_class"].strip(): - raise ContractError("goal and task_class must be non-empty strings") - if not isinstance(value["acceptance"], list) or not value["acceptance"]: - raise ContractError("acceptance must be non-empty") - for item in value["acceptance"]: - _exact(item, {"id", "description", "evidence_kind"}, {"id", "description", "evidence_kind"}, "acceptance item") - if item["evidence_kind"] not in {"review", "check", "artifact", "host"}: - raise ContractError("invalid evidence_kind") - if not isinstance(value["checks"], list): raise ContractError("checks must be an array") - for check in value["checks"]: - _exact(check, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, "check") - if not isinstance(check["argv"], list) or not check["argv"] or not all(isinstance(v, str) and v for v in check["argv"]): raise ContractError("check argv must be a non-empty string array") - _relative(check["cwd"], "check cwd") - if not isinstance(check["timeout_seconds"], int) or isinstance(check["timeout_seconds"], bool) or check["timeout_seconds"] <= 0: raise ContractError("check timeout must be positive") - scope = value["scope"]; _exact(scope, {"read_paths", "write_paths"}, {"read_paths", "write_paths"}, "scope") - for p in scope["read_paths"] + scope["write_paths"]: _relative(p, "scope path") - if value["workflow"] == "branch-review" and scope["write_paths"]: raise ContractError("branch review cannot write") - budget = value["budget"]; required = {"wall_seconds", "max_worker_invocations", "max_revisions", "max_fallbacks_per_step"}; _exact(budget, required, required, "budget") - for key, number in budget.items(): - if not isinstance(number, int) or isinstance(number, bool) or number < 0: raise ContractError(f"budget {key} must be a finite non-negative integer") - if budget["wall_seconds"] == 0 or budget["max_worker_invocations"] == 0: raise ContractError("wall_seconds and max_worker_invocations must be positive") From 39c0379e03f5adf0b4d98934f648061f30ca8cc0 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 6 Sep 2026 18:35:47 +0530 Subject: [PATCH 009/197] fix: version native Codex launch preparation --- plugin/core/src/devsquad/adapters.py | 20 ++++++++++++++++++++ test/core/test_m1.py | 13 ++++++++++++- 2 files changed, 32 insertions(+), 1 deletion(-) diff --git a/plugin/core/src/devsquad/adapters.py b/plugin/core/src/devsquad/adapters.py index 70d2bcd..ea8bdd9 100644 --- a/plugin/core/src/devsquad/adapters.py +++ b/plugin/core/src/devsquad/adapters.py @@ -51,6 +51,9 @@ def load(cls, path: Path) -> "AdapterManifest": def resolve_binary(self) -> str | None: return next((p for name in self.binary_candidates if (p := shutil.which(name))), None) + def with_model_efforts(self, mapping: dict[str, tuple[str, ...]]) -> "AdapterManifest": + return AdapterManifest(self.name, self.transport, self.binary_candidates, self.model_provider, mapping, self.permission_profiles, self.output_format, self.verified_versions) + def _permission_args(manifest: AdapterManifest, permission: str) -> tuple[str, ...]: try: @@ -102,6 +105,23 @@ def prepare_cli( return LaunchSpec(SCHEMA_VERSION, manifest.name, "cli_exec", tuple(args), str(Path(cwd).resolve()), stdin_path, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) +def prepare_native_codex(manifest: AdapterManifest, *, cwd: str, model: str, effort: str, permission: str, timeout_seconds: int, harness_version_value: str) -> LaunchSpec: + """Prepare the supervisor-owned app-server child without spawning it.""" + if manifest.name != "codex" or manifest.transport != "native_protocol": + raise ContractError("native Codex manifest required") + binary = manifest.resolve_binary() + if not binary: + raise ContractError("adapter unavailable: codex") + if harness_version_value not in manifest.verified_versions: + raise ProfileUnsupported(f"unverified Codex app-server version: {harness_version_value}") + supported = manifest.efforts_by_model.get(model) + if supported is None or effort not in supported: + raise ProfileUnsupported(f"unsupported or unverified effort {effort!r} for codex model {model!r}") + _permission_args(manifest, permission) + requested = ExecutionIdentity("codex", harness_version_value, "openai", None, model, effort, (), permission, None, "verified") + return LaunchSpec(SCHEMA_VERSION, "codex", "native_protocol", (binary, "app-server", "--listen", "stdio://"), str(Path(cwd).resolve()), None, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) + + def _provider_records(adapter: str, stdout: str) -> tuple[list[dict[str, Any]], bool, str | None]: records: list[dict[str, Any]] = [] deliverable = False diff --git a/test/core/test_m1.py b/test/core/test_m1.py index 2fe7341..93bc16f 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -12,7 +12,7 @@ import sys sys.path.insert(0, str(CORE / "src")) -from devsquad.adapters import AdapterManifest, classify_cli, prepare_cli +from devsquad.adapters import AdapterManifest, classify_cli, prepare_cli, prepare_native_codex from devsquad.catalog import update_last_good from devsquad.codex_protocol import NativeTurnState, model_list_request, parse_model_page from devsquad.contracts import ContractError, validate_launch_payload @@ -84,6 +84,17 @@ def test_codex_real_jsonl_shape_requires_agent_message(self): startup = json.dumps({"type":"thread.started","thread_id":"x"}) + "\n" + json.dumps({"type":"turn.completed","usage":{}}) self.assertEqual(classify_cli(spec, returncode=0, stdout=startup, stderr="").execution_status, "malformed") + def test_native_launch_is_preparation_only_and_version_scoped(self): + temp, binary = self.fake_path("codex"); self.addCleanup(temp.cleanup) + manifest = self.manifest("codex").with_model_efforts({"gpt-test": ("low",)}) + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.135.0") + self.assertEqual(spec.transport, "native_protocol") + self.assertEqual(spec.argv[-2:], ("--listen", "stdio://")) + self.assertEqual(spec.environment["DEVSQUAD_WORKER"], "1") + with self.assertRaises(ContractError): + prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli future") + def test_launch_round_trip_is_strict(self): temp, binary = self.fake_path("grok"); self.addCleanup(temp.cleanup) with patch.dict(os.environ, {"PATH": str(binary.parent)}): From 1086b8a4db21ea29f3f13c7ea1abb1fab11b431f Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 00:49:26 +0530 Subject: [PATCH 010/197] fix: close M1 invocation gate gaps --- plugin/core/adapters/antigravity/adapter.json | 2 +- plugin/core/adapters/grok/adapter.json | 2 +- plugin/core/schemas/policy.schema.json | 14 +- plugin/core/schemas/task.schema.json | 22 ++- plugin/core/src/devsquad/adapters.py | 19 ++- plugin/core/src/devsquad/catalog.py | 12 ++ plugin/core/src/devsquad/cli.py | 17 +- plugin/core/src/devsquad/codex_protocol.py | 154 +++++++++++++++++- plugin/core/src/devsquad/contracts.py | 13 ++ plugin/core/src/devsquad/validation.py | 38 +++++ plugin/lib/adapter.sh | 27 ++- plugin/lib/gemini-wrapper.sh | 11 +- test/core/fakes/codex_app_server.py | 20 +++ test/core/test_m1.py | 94 ++++++++++- test/test_m1_legacy.sh | 69 +++++++- test/test_wrapper_contract.sh | 7 + 16 files changed, 472 insertions(+), 49 deletions(-) create mode 100644 test/core/fakes/codex_app_server.py diff --git a/plugin/core/adapters/antigravity/adapter.json b/plugin/core/adapters/antigravity/adapter.json index d97db19..d95a2c7 100644 --- a/plugin/core/adapters/antigravity/adapter.json +++ b/plugin/core/adapters/antigravity/adapter.json @@ -5,6 +5,6 @@ "binary_candidates": ["agy", "antigravity"], "model_provider": null, "capabilities": {"efforts_by_model": {}, "native_model_list": false, "resume": true}, - "permission_profiles": {"read_only": ["--sandbox"], "workspace_write": []}, + "permission_profiles": {"read_only": ["--sandbox"], "workspace_write": ["--sandbox"]}, "output_format": "json" } diff --git a/plugin/core/adapters/grok/adapter.json b/plugin/core/adapters/grok/adapter.json index 94f8162..3daabdd 100644 --- a/plugin/core/adapters/grok/adapter.json +++ b/plugin/core/adapters/grok/adapter.json @@ -5,6 +5,6 @@ "binary_candidates": ["grok"], "model_provider": "xai", "capabilities": {"efforts_by_model": {}, "native_model_list": false, "resume": true}, - "permission_profiles": {"read_only": ["--disable-web-search"], "workspace_write": []}, + "permission_profiles": {"read_only": ["--disable-web-search"]}, "output_format": "json" } diff --git a/plugin/core/schemas/policy.schema.json b/plugin/core/schemas/policy.schema.json index 9541070..3e03ae4 100644 --- a/plugin/core/schemas/policy.schema.json +++ b/plugin/core/schemas/policy.schema.json @@ -1,6 +1,12 @@ { - "$schema": "https://json-schema.org/draft/2020-12/schema", "$id": "https://devsquad.local/schemas/policy-v1.json", - "type": "object", "additionalProperties": false, - "required": ["schema_version", "id", "version", "roles", "task_classes", "require_different_model_for_review", "account_pools", "experiment_budget"], - "properties": {"schema_version": {"const": 1}, "id": {"type": "string", "minLength": 1}, "version": {"type": "integer", "minimum": 1}, "roles": {"type": "object"}, "task_classes": {"type": "object"}, "require_different_model_for_review": {"type": "boolean"}, "prefer_different_harness_for_review": {"type": "boolean"}, "account_pools": {"type": "object"}, "experiment_budget": {"type": "object"}} + "$schema":"https://json-schema.org/draft/2020-12/schema","$id":"https://devsquad.local/schemas/policy-v1.json", + "type":"object","additionalProperties":false, + "required":["schema_version","id","version","roles","task_classes","require_different_model_for_review","account_pools","experiment_budget"], + "properties":{ + "schema_version":{"const":1},"id":{"type":"string","minLength":1},"version":{"type":"integer","minimum":1}, + "roles":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"type":"array","minItems":1,"items":{"$ref":"#/$defs/candidate"}}}, + "task_classes":{"type":"object"},"require_different_model_for_review":{"type":"boolean"},"prefer_different_harness_for_review":{"type":"boolean"}, + "account_pools":{"type":"object"},"experiment_budget":{"type":"object"} + }, + "$defs":{"candidate":{"type":"object","additionalProperties":false,"required":["kind","id"],"properties":{"kind":{"enum":["profile","alias"]},"id":{"type":"string","minLength":1}}}} } diff --git a/plugin/core/schemas/task.schema.json b/plugin/core/schemas/task.schema.json index 01207f5..d452a8b 100644 --- a/plugin/core/schemas/task.schema.json +++ b/plugin/core/schemas/task.schema.json @@ -3,9 +3,23 @@ "type": "object", "additionalProperties": false, "required": ["schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin"], "properties": { - "schema_version": {"const": 1}, - "project": {"type": "object", "additionalProperties": false, "required": ["repo_path", "base_ref", "target_ref"], "properties": {"repo_path": {"type": "string"}, "base_ref": {"type": "string"}, "target_ref": {"type": "string"}}}, - "workflow": {"enum": ["branch-review", "issue-delivery"]}, "goal": {"type": "string", "minLength": 1}, "task_class": {"type": "string", "minLength": 1}, - "acceptance": {"type": "array", "minItems": 1}, "checks": {"type": "array"}, "scope": {"type": "object"}, "lead": {"type": "object"}, "routing": {"type": "object"}, "budget": {"type": "object"}, "origin": {"type": "object"} + "schema_version": {"const": 1}, "workflow": {"enum": ["branch-review", "issue-delivery"]}, + "goal": {"type": "string", "minLength": 1}, "task_class": {"type": "string", "minLength": 1}, + "project": {"$ref": "#/$defs/project"}, "acceptance": {"type": "array", "minItems": 1, "items": {"$ref": "#/$defs/acceptance"}}, + "checks": {"type": "array", "items": {"$ref": "#/$defs/check"}}, "scope": {"$ref": "#/$defs/scope"}, + "lead": {"$ref": "#/$defs/lead"}, "routing": {"$ref": "#/$defs/routing"}, "budget": {"$ref": "#/$defs/budget"}, + "origin": {"$ref": "#/$defs/origin"}, "review": {"$ref": "#/$defs/review"} + }, + "$defs": { + "project": {"type":"object","additionalProperties":false,"required":["repo_path","base_ref","target_ref"],"properties":{"repo_path":{"type":"string","pattern":"^/"},"base_ref":{"type":"string","minLength":1},"target_ref":{"type":"string","minLength":1}}}, + "acceptance": {"type":"object","additionalProperties":false,"required":["id","description","evidence_kind"],"properties":{"id":{"type":"string","minLength":1},"description":{"type":"string","minLength":1},"evidence_kind":{"enum":["review","check","artifact","host"]}}}, + "check": {"type":"object","additionalProperties":false,"required":["id","argv","cwd","timeout_seconds","required_to_pass"],"properties":{"id":{"type":"string","minLength":1},"argv":{"type":"array","minItems":1,"items":{"type":"string","minLength":1}},"cwd":{"type":"string"},"timeout_seconds":{"type":"integer","minimum":1},"required_to_pass":{"type":"boolean"}}}, + "scope": {"type":"object","additionalProperties":false,"required":["read_paths","write_paths"],"properties":{"read_paths":{"type":"array","items":{"type":"string"}},"write_paths":{"type":"array","items":{"type":"string"}}}}, + "lead": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["host","headless"]}}}, + "override": {"type":"object","additionalProperties":false,"required":["profile_id"],"properties":{"profile_id":{"type":"string","minLength":1},"fallback":{"enum":["none","policy"]}}}, + "routing": {"type":"object","additionalProperties":false,"required":["profiles_file","policy_file"],"properties":{"profiles_file":{"type":"string","minLength":1},"policy_file":{"type":"string","minLength":1},"overrides":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"$ref":"#/$defs/override"}}}}, + "budget": {"type":"object","additionalProperties":false,"required":["wall_seconds","max_worker_invocations","max_revisions","max_fallbacks_per_step"],"properties":{"wall_seconds":{"type":"integer","minimum":1},"max_worker_invocations":{"type":"integer","minimum":1},"max_revisions":{"type":"integer","minimum":0},"max_fallbacks_per_step":{"type":"integer","minimum":0}}}, + "origin": {"type":"object","additionalProperties":false,"required":["surface"],"properties":{"surface":{"type":"string","minLength":1},"session_ref":{"type":"string"}}}, + "review": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["standard","adversarial"]},"focus":{"type":"string","minLength":1}}} } } diff --git a/plugin/core/src/devsquad/adapters.py b/plugin/core/src/devsquad/adapters.py index ea8bdd9..c493a32 100644 --- a/plugin/core/src/devsquad/adapters.py +++ b/plugin/core/src/devsquad/adapters.py @@ -12,6 +12,7 @@ from typing import Any from .contracts import ContractError, ExecutionIdentity, LaunchSpec, NormalizedResult, ProfileUnsupported, SCHEMA_VERSION +from .catalog import verified_efforts ERROR_PATTERNS = ( ("AUTH_ERROR", re.compile(r"auth|unauthorized|ineligible|\b401\b|\b403\b", re.I)), @@ -122,10 +123,16 @@ def prepare_native_codex(manifest: AdapterManifest, *, cwd: str, model: str, eff return LaunchSpec(SCHEMA_VERSION, "codex", "native_protocol", (binary, "app-server", "--listen", "stdio://"), str(Path(cwd).resolve()), None, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) -def _provider_records(adapter: str, stdout: str) -> tuple[list[dict[str, Any]], bool, str | None]: +def prepare_native_codex_from_catalog(manifest: AdapterManifest, snapshot: dict[str, Any], *, cwd: str, model: str, effort: str, permission: str, timeout_seconds: int, harness_version_value: str) -> LaunchSpec: + efforts = verified_efforts(snapshot, harness="codex", version=harness_version_value, model_id=model) + return prepare_native_codex(manifest.with_model_efforts({model: efforts}), cwd=cwd, model=model, effort=effort, permission=permission, timeout_seconds=timeout_seconds, harness_version_value=harness_version_value) + + +def _provider_records(adapter: str, stdout: str) -> tuple[list[dict[str, Any]], bool, bool, str | None]: records: list[dict[str, Any]] = [] deliverable = False error_text = None + terminal = False try: document = json.loads(stdout) source = document if isinstance(document, list) else [document] @@ -133,11 +140,13 @@ def _provider_records(adapter: str, stdout: str) -> tuple[list[dict[str, Any]], source = [json.loads(line) for line in stdout.splitlines() if line.strip()] for item in source: if not isinstance(item, dict): - raise json.JSONDecodeError("record is not an object", line, 0) + raise json.JSONDecodeError("record is not an object", stdout, 0) records.append(item) if item.get("is_error") is True or item.get("error"): error_text = str(item.get("error") or item.get("result") or item.get("message")) kind = item.get("type") + if kind in {"result", "turn.completed"}: + terminal = True if adapter == "codex" and kind == "item.completed": native_item = item.get("item") or {} if native_item.get("type") in {"agent_message", "agentMessage"}: @@ -147,7 +156,7 @@ def _provider_records(adapter: str, stdout: str) -> tuple[list[dict[str, Any]], payload = item.get("result") or item.get("message") or item.get("text") if isinstance(payload, str) and payload.strip(): deliverable = True - return records, deliverable, error_text + return records, deliverable, terminal, error_text def classify_cli(spec: LaunchSpec, *, returncode: int, stdout: str, stderr: str, timed_out: bool = False) -> NormalizedResult: @@ -163,11 +172,11 @@ def classify_cli(spec: LaunchSpec, *, returncode: int, stdout: str, stderr: str, status, code = "malformed", "CLI_ERROR" elif spec.adapter in {"codex", "antigravity", "grok"}: try: - _, deliverable, provider_error = _provider_records(spec.adapter, stdout) + _, deliverable, terminal, provider_error = _provider_records(spec.adapter, stdout) if provider_error: code = next((candidate for candidate, pattern in ERROR_PATTERNS if pattern.search(provider_error)), "CLI_ERROR") status = "denied" if re.search(r"permission denied|tool (?:use )?denied|not allowed", provider_error, re.I) else "failed" - elif not deliverable: + elif not deliverable or not terminal: status, code = "malformed", "CLI_ERROR" except json.JSONDecodeError: status, code = "malformed", "CLI_ERROR" diff --git a/plugin/core/src/devsquad/catalog.py b/plugin/core/src/devsquad/catalog.py index 59bc94c..5e8f38a 100644 --- a/plugin/core/src/devsquad/catalog.py +++ b/plugin/core/src/devsquad/catalog.py @@ -58,3 +58,15 @@ def _atomic_json(path: Path, value: dict[str, Any]) -> None: tmp = path.with_name(f"{path.name}.tmp.{os.getpid()}") tmp.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") os.replace(tmp, path) + + +def verified_efforts(snapshot: dict[str, Any], *, harness: str, version: str, model_id: str) -> tuple[str, ...]: + if snapshot.get("complete") is not True or snapshot.get("harness") != harness or snapshot.get("harness_version") != version: + raise ContractError("catalog snapshot does not verify this harness version") + matches = [m for m in snapshot.get("models", []) if m.get("id") == model_id] + if len(matches) != 1: + raise ContractError(f"model is not uniquely present in verified catalog: {model_id}") + efforts = matches[0].get("supported_efforts") + if not isinstance(efforts, list) or not all(isinstance(v, str) for v in efforts): + raise ContractError("model effort metadata is unknown") + return tuple(efforts) diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 565c647..5ca4ab4 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -8,7 +8,7 @@ from pathlib import Path from . import __version__ -from .adapters import AdapterManifest, classify_cli, harness_version, prepare_cli +from .adapters import AdapterManifest, classify_cli, harness_version, prepare_cli, prepare_native_codex_from_catalog from .contracts import ContractError, envelope, error_payload SOURCE_ROOT = Path(__file__).resolve().parents[2] @@ -32,7 +32,18 @@ def command_doctor(_: argparse.Namespace) -> tuple[dict, int]: def command_prepare(args: argparse.Namespace) -> dict: manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") - spec = prepare_cli(manifest, prompt=args.prompt, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) + transport = args.transport or manifest.transport + if transport == "native_protocol": + if manifest.name != "codex" or not args.catalog_file or not args.model or not args.effort: + raise ContractError("native preparation requires Codex, --catalog-file, --model and --effort") + binary = manifest.resolve_binary() + version = harness_version(binary) if binary else None + snapshot = json.loads(Path(args.catalog_file).read_text()) + spec = prepare_native_codex_from_catalog(manifest, snapshot, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout, harness_version_value=version or "unknown") + elif transport == "cli_exec": + spec = prepare_cli(manifest, prompt=args.prompt, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) + else: + raise ContractError(f"unsupported transport: {transport}") return envelope(data=spec.to_dict()), 0 @@ -61,6 +72,8 @@ def parser() -> argparse.ArgumentParser: cmd.add_argument("--effort") cmd.add_argument("--permission", choices=("read_only", "workspace_write"), default="read_only") cmd.add_argument("--timeout", type=int, default=90) + cmd.add_argument("--transport", choices=("cli_exec", "native_protocol")) + cmd.add_argument("--catalog-file") if name == "prepare": cmd.add_argument("--prompt", required=True) else: diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index edfc56e..d975f60 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -7,11 +7,57 @@ from __future__ import annotations from dataclasses import dataclass, field +import json +import os +import selectors +import time from typing import Any from .contracts import ContractError +class JsonLinePeer: + """Protocol codec over supervisor-owned streams; does not own the process.""" + def __init__(self, reader: Any, writer: Any, *, max_frame_bytes: int = 4 * 1024 * 1024): + self.reader, self.writer = reader, writer + self._fd = reader.fileno() + self._buffer = bytearray() + self._max_frame_bytes = max_frame_bytes + + def send(self, message: dict[str, Any]) -> None: + self.writer.write(json.dumps(message, separators=(",", ":")) + "\n") + self.writer.flush() + + def receive(self, timeout_seconds: float) -> dict[str, Any]: + deadline = time.monotonic() + timeout_seconds + while b"\n" not in self._buffer: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("native protocol response timed out") + selector = selectors.DefaultSelector() + try: + selector.register(self._fd, selectors.EVENT_READ) + if not selector.select(remaining): + raise TimeoutError("native protocol response timed out") + finally: + selector.close() + chunk = os.read(self._fd, 65536) + if not chunk: + raise EOFError("native protocol disconnected") + self._buffer.extend(chunk) + if len(self._buffer) > self._max_frame_bytes: + raise ContractError("native protocol frame exceeds byte bound") + raw, _, remainder = self._buffer.partition(b"\n") + self._buffer = bytearray(remainder) + try: + value = json.loads(raw.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("native protocol returned malformed JSON") from exc + if not isinstance(value, dict): + raise ContractError("native protocol message must be an object") + return value + + def request(request_id: int, method: str, params: dict[str, Any] | None = None) -> dict[str, Any]: return {"id": request_id, "method": method, "params": params or {}} @@ -20,6 +66,13 @@ def initialize_request(request_id: int = 1) -> dict[str, Any]: return request(request_id, "initialize", {"clientInfo": {"name": "devsquad", "version": "0.1.0"}, "capabilities": {"experimentalApi": True}}) +def thread_start_request(request_id: int, *, cwd: str, model: str, permission: str) -> dict[str, Any]: + sandbox = {"read_only": "read-only", "workspace_write": "workspace-write"}.get(permission) + if sandbox is None: + raise ContractError(f"unsupported native permission: {permission}") + return request(request_id, "thread/start", {"cwd": cwd, "model": model, "sandbox": sandbox, "approvalPolicy": "never", "ephemeral": False}) + + def model_list_request(request_id: int, cursor: str | None = None, limit: int = 100) -> dict[str, Any]: params: dict[str, Any] = {"limit": limit} if cursor: @@ -27,15 +80,31 @@ def model_list_request(request_id: int, cursor: str | None = None, limit: int = return request(request_id, "model/list", params) -def turn_start_request(request_id: int, *, thread_id: str, prompt: str, model: str | None, effort: str | None, cwd: str, sandbox: str) -> dict[str, Any]: - params: dict[str, Any] = {"threadId": thread_id, "input": [{"type": "text", "text": prompt}], "cwd": cwd, "permissions": sandbox} +def turn_start_request(request_id: int, *, thread_id: str, prompt: str, model: str | None, effort: str | None, cwd: str, permission: str, output_schema: dict[str, Any] | None = None) -> dict[str, Any]: + policy = {"read_only": {"type": "readOnly", "networkAccess": False}, "workspace_write": {"type": "workspaceWrite", "writableRoots": [cwd], "networkAccess": False}}.get(permission) + if policy is None: + raise ContractError(f"unsupported native permission: {permission}") + params: dict[str, Any] = {"threadId": thread_id, "input": [{"type": "text", "text": prompt}], "cwd": cwd, "sandboxPolicy": policy, "approvalPolicy": "never"} if model: params["model"] = model if effort: params["effort"] = effort + if output_schema is not None: + params["outputSchema"] = output_schema return request(request_id, "turn/start", params) +def review_start_request(request_id: int, *, thread_id: str, target: dict[str, Any]) -> dict[str, Any]: + allowed = {"uncommittedChanges", "baseBranch", "commit", "custom"} + if target.get("type") not in allowed: + raise ContractError("unsupported review target") + return request(request_id, "review/start", {"threadId": thread_id, "target": target, "delivery": "inline"}) + + +def turn_interrupt_request(request_id: int, *, thread_id: str, turn_id: str) -> dict[str, Any]: + return request(request_id, "turn/interrupt", {"threadId": thread_id, "turnId": turn_id}) + + @dataclass class NativeTurnState: thread_id: str | None = None @@ -44,6 +113,8 @@ class NativeTurnState: interrupted_acknowledged: bool = False output: list[str] = field(default_factory=list) events: list[dict[str, Any]] = field(default_factory=list) + terminal_status: str | None = None + error: dict[str, Any] | None = None def consume(self, message: dict[str, Any]) -> None: if not isinstance(message, dict): @@ -53,14 +124,34 @@ def consume(self, message: dict[str, Any]) -> None: params = message.get("params") or message.get("result") or {} if method in {"thread/started", "thread/start/completed"}: self.thread_id = params.get("thread", {}).get("id") or params.get("threadId") or self.thread_id - if method in {"turn/started", "turn/start/completed"}: - self.turn_id = params.get("turn", {}).get("id") or params.get("turnId") or self.turn_id + message_thread = params.get("threadId") + turn = params.get("turn") or {} + message_turn = turn.get("id") or params.get("turnId") + if self.thread_id and message_thread and message_thread != self.thread_id: + return + if self.turn_id and message_turn and message_turn != self.turn_id: + return + if method == "turn/started": + self.thread_id = message_thread or self.thread_id + self.turn_id = message_turn or self.turn_id if method in {"item/agentMessage/delta", "turn/output/delta"}: self.output.append(params.get("delta", "")) - if method in {"turn/interrupt/completed", "turn/interrupted/acknowledged"}: - self.interrupted_acknowledged = True - if method in {"turn/completed", "review/completed", "turn/failed", "turn/interrupted"}: + if method == "turn/completed" and self.turn_id and message_turn == self.turn_id: + self.terminal = True + self.terminal_status = turn.get("status") + if method == "error" and self.turn_id and message_turn == self.turn_id and not params.get("willRetry", False): self.terminal = True + self.terminal_status = "failed" + self.error = params.get("error") + + def acknowledge_interrupt(self, response: dict[str, Any]) -> None: + if "error" in response: + raise ContractError(f"turn/interrupt failed: {response['error']}") + self.interrupted_acknowledged = True + + def disconnected(self) -> None: + if not self.terminal: + self.terminal_status = "transport_disconnected" def parse_model_page(response: dict[str, Any]) -> tuple[list[dict[str, Any]], str | None]: @@ -69,8 +160,55 @@ def parse_model_page(response: dict[str, Any]) -> tuple[list[dict[str, Any]], st result = response.get("result") if not isinstance(result, dict): raise ContractError("model/list missing result") - models = result.get("data") or result.get("models") + models = result["data"] if "data" in result else result.get("models") if not isinstance(models, list): raise ContractError("model/list is incomplete") cursor = result.get("nextCursor") or result.get("next_cursor") return models, cursor + + +def collect_model_pages(fetch_page: Any, *, max_pages: int = 100) -> list[dict[str, Any]]: + cursor = None + seen: set[str] = set() + all_models: list[dict[str, Any]] = [] + for _ in range(max_pages): + models, next_cursor = parse_model_page(fetch_page(cursor)) + all_models.extend(models) + if next_cursor is None: + return all_models + if next_cursor in seen: + raise ContractError("model/list repeated pagination cursor") + seen.add(next_cursor) + cursor = next_cursor + raise ContractError("model/list exceeded page bound") + + +def receive_response(peer: JsonLinePeer, request_id: int, *, timeout_seconds: float, on_notification: Any | None = None) -> dict[str, Any]: + deadline = time.monotonic() + timeout_seconds + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError(f"native request {request_id} timed out") + message = peer.receive(remaining) + if message.get("id") == request_id: + return message + if "method" in message and "id" not in message: + if on_notification: + on_notification(message) + continue + raise ContractError(f"unexpected native response while awaiting request {request_id}") + + +def discover_models(peer: JsonLinePeer, *, first_request_id: int = 10, timeout_seconds: float = 5, max_pages: int = 100) -> list[dict[str, Any]]: + """Collect a complete native snapshot from an already initialized peer.""" + request_id, cursor = first_request_id, None + responses: list[dict[str, Any]] = [] + for _ in range(max_pages): + peer.send(model_list_request(request_id, cursor)) + response = receive_response(peer, request_id, timeout_seconds=timeout_seconds) + responses.append(response) + _, cursor = parse_model_page(response) + if cursor is None: + return collect_model_pages(lambda ignored: responses.pop(0), max_pages=len(responses)) + request_id += 1 + raise ContractError("model/list exceeded page bound") diff --git a/plugin/core/src/devsquad/contracts.py b/plugin/core/src/devsquad/contracts.py index 9fca7dc..dd8f584 100644 --- a/plugin/core/src/devsquad/contracts.py +++ b/plugin/core/src/devsquad/contracts.py @@ -34,6 +34,16 @@ class ExecutionIdentity: account_pool: str | None = None verification: Verification = "unknown" + def __post_init__(self) -> None: + if not isinstance(self.harness, str) or not self.harness: + raise ContractError("identity harness must be non-empty") + if self.permissions not in {"read_only", "workspace_write"}: + raise ContractError("identity permission is invalid") + if self.verification not in {"verified", "unverified", "unavailable", "unknown"}: + raise ContractError("identity verification is invalid") + if not isinstance(self.tools, tuple) or not all(isinstance(v, str) for v in self.tools): + raise ContractError("identity tools must be a string tuple") + @dataclass(frozen=True) class LaunchSpec: @@ -60,6 +70,9 @@ def __post_init__(self) -> None: raise ContractError("timeout_seconds must be positive") if not Path(self.cwd).is_absolute(): raise ContractError("cwd must be absolute") + allowed_env = {"DEVSQUAD_WORKER", "DEVSQUAD_RUN_ID", "DEVSQUAD_ATTEMPT_ID", "DEVSQUAD_DELEGATION_DEPTH"} + if not isinstance(self.environment, dict) or set(self.environment) - allowed_env or not all(isinstance(k, str) and isinstance(v, str) for k, v in self.environment.items()): + raise ContractError("environment contains non-allowlisted or non-string values") def to_dict(self) -> dict[str, Any]: return asdict(self) diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py index 3be2dc9..4724aff 100644 --- a/plugin/core/src/devsquad/validation.py +++ b/plugin/core/src/devsquad/validation.py @@ -54,6 +54,13 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) routing = value["routing"]; _exact(routing, {"profiles_file", "policy_file", "overrides"}, {"profiles_file", "policy_file"}, "routing") for key in ("profiles_file", "policy_file"): if not isinstance(routing[key], str) or not routing[key]: raise ContractError(f"routing {key} must be a path") + overrides = routing.get("overrides", {}) + if not isinstance(overrides, dict): raise ContractError("routing overrides must be an object") + for role, override in overrides.items(): + if role not in {"implementer", "reviewer", "lead", "researcher"}: raise ContractError("invalid override role") + _exact(override, {"profile_id", "fallback"}, {"profile_id"}, "routing override") + if not isinstance(override["profile_id"], str) or not override["profile_id"]: raise ContractError("override profile_id must be non-empty") + if override.get("fallback", "none") not in {"none", "policy"}: raise ContractError("override fallback must be none or policy") origin = value["origin"]; _exact(origin, {"surface", "session_ref"}, {"surface"}, "origin") if not isinstance(origin["surface"], str) or not origin["surface"]: raise ContractError("origin surface must be non-empty") if "review" in value: @@ -64,3 +71,34 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) for key, number in budget.items(): if not isinstance(number, int) or isinstance(number, bool) or number < 0: raise ContractError(f"budget {key} must be a finite non-negative integer") if budget["wall_seconds"] == 0 or budget["max_worker_invocations"] == 0: raise ContractError("wall_seconds and max_worker_invocations must be positive") + + +def validate_profile(value: dict[str, Any]) -> None: + fields = {"id", "harness", "model_family", "model_id", "effort", "required_tools", "permission_policy", "account_pool_id", "billing_mode", "quality_status", "evidence_refs"} + _exact(value, fields, fields, "profile") + for key in ("id", "harness", "model_family", "model_id", "account_pool_id"): + if not isinstance(value[key], str) or not value[key]: raise ContractError(f"profile {key} must be non-empty") + effort = value["effort"]; _exact(effort, {"value", "transport"}, {"value", "transport"}, "profile effort") + if effort["value"] is not None and not isinstance(effort["value"], str): raise ContractError("effort value must be string or null") + if effort["transport"] not in {"native", "model_variant", "provider_default"}: raise ContractError("invalid effort transport") + for key in ("required_tools", "evidence_refs"): + if not isinstance(value[key], list) or not all(isinstance(v, str) for v in value[key]) or len(set(value[key])) != len(value[key]): raise ContractError(f"profile {key} must contain unique strings") + if value["permission_policy"] not in {"read_only", "workspace_write"}: raise ContractError("invalid permission policy") + if value["billing_mode"] not in {"subscription", "paid_api"}: raise ContractError("invalid billing mode") + if value["quality_status"] not in {"unvalidated", "trial", "proven", "suspended"}: raise ContractError("invalid quality status") + + +def validate_policy(value: dict[str, Any]) -> None: + fields = {"schema_version", "id", "version", "roles", "task_classes", "require_different_model_for_review", "prefer_different_harness_for_review", "account_pools", "experiment_budget"} + required = fields - {"prefer_different_harness_for_review"} + _exact(value, fields, required, "policy") + if type(value["schema_version"]) is not int or value["schema_version"] != 1 or type(value["version"]) is not int or value["version"] < 1: raise ContractError("invalid policy version") + if type(value["require_different_model_for_review"]) is not bool or ("prefer_different_harness_for_review" in value and type(value["prefer_different_harness_for_review"]) is not bool): raise ContractError("policy review flags must be boolean") + if not isinstance(value["roles"], dict) or set(value["roles"]) - {"implementer", "reviewer", "lead", "researcher"}: raise ContractError("invalid policy roles") + for candidates in value["roles"].values(): + if not isinstance(candidates, list) or not candidates: raise ContractError("role candidates must be non-empty arrays") + for ref in candidates: + _exact(ref, {"kind", "id"}, {"kind", "id"}, "candidate reference") + if ref["kind"] not in {"profile", "alias"} or not isinstance(ref["id"], str) or not ref["id"]: raise ContractError("invalid candidate reference") + for key in ("task_classes", "account_pools", "experiment_budget"): + if not isinstance(value[key], dict): raise ContractError(f"policy {key} must be an object") diff --git a/plugin/lib/adapter.sh b/plugin/lib/adapter.sh index 6679766..45dba9b 100644 --- a/plugin/lib/adapter.sh +++ b/plugin/lib/adapter.sh @@ -118,6 +118,12 @@ _adapter_invoke() { local cli cli=$(_adapter_resolve_cli) + if [[ -n "${DEVSQUAD_TEST_ADAPTER_EXECUTABLE:-}" ]]; then + case "$DEVSQUAD_TEST_ADAPTER_EXECUTABLE" in + /*) [[ -x "$DEVSQUAD_TEST_ADAPTER_EXECUTABLE" ]] && cli="$DEVSQUAD_TEST_ADAPTER_EXECUTABLE" ;; + *) _adapter_fail "CLI_ERROR: test adapter executable must be absolute"; return 1 ;; + esac + fi if [[ -z "$cli" ]]; then _adapter_fail "CLI_ERROR: ${ADAPTER_MISSING_MSG}" return 1 @@ -129,8 +135,8 @@ _adapter_invoke() { _adapter_build_args "$final_prompt" "$model" "$timeout_secs" local timeout_cmd="" - if command -v timeout &>/dev/null; then timeout_cmd="timeout" - elif command -v gtimeout &>/dev/null; then timeout_cmd="gtimeout" + if [[ "${DEVSQUAD_FORCE_PORTABLE_TIMEOUT:-0}" != "1" ]] && command -v timeout &>/dev/null; then timeout_cmd="timeout" + elif [[ "${DEVSQUAD_FORCE_PORTABLE_TIMEOUT:-0}" != "1" ]] && command -v gtimeout &>/dev/null; then timeout_cmd="gtimeout" fi local stderr_file stdout_file @@ -157,24 +163,28 @@ _adapter_invoke() { local cli_pid=$! local timed_out_file="${stdout_file}.timed-out" local process_snapshot="${stdout_file}.processes" + local watchdog_snapshot="${stdout_file}.watchdog-processes" # Redirect the watchdog itself: inherited capture descriptors were the # reason a successful immediate command waited for the full timeout. - ( sleep "$timeout_secs"; _adapter_snapshot_tree "$cli_pid" > "$process_snapshot"; : > "$timed_out_file"; _adapter_signal_snapshot "$process_snapshot" TERM ) >/dev/null 2>&1 & + ( sleep "$timeout_secs"; _adapter_snapshot_tree "$cli_pid" > "$process_snapshot"; : > "$timed_out_file"; _adapter_signal_snapshot "$process_snapshot" TERM; sleep 0.1; _adapter_signal_snapshot "$process_snapshot" KILL ) >/dev/null 2>&1 & local watchdog_pid=$! if wait "$cli_pid"; then exit_code=0 else exit_code=$? fi - kill "$watchdog_pid" 2>/dev/null || true + # Snapshot the watchdog subtree while its sleep child is still attached, + # then stop the exact identities. This never scans after orphaning. + _adapter_snapshot_tree "$watchdog_pid" > "$watchdog_snapshot" + _adapter_signal_snapshot "$watchdog_snapshot" KILL wait "$watchdog_pid" 2>/dev/null || true if [[ -f "$timed_out_file" ]]; then - # Give descendants a brief grace, then ensure stubborn children vanish. - sleep 0.1 + # The watchdog performs escalation before wait can return. Repeat the + # exact captured set defensively; never discover unrelated PIDs here. _adapter_signal_snapshot "$process_snapshot" KILL exit_code=124 fi - rm -f "$timed_out_file" "$process_snapshot" + rm -f "$timed_out_file" "$process_snapshot" "$watchdog_snapshot" fi local stdout stderr_content @@ -187,7 +197,8 @@ _adapter_invoke() { _adapter_fail "AUTH_ERROR: ${agent} CLI is not authenticated. ${ADAPTER_AUTH_HINT}" elif [[ $exit_code -eq 0 ]]; then if [[ -z "$stdout" ]]; then - echo "WARNING: ${agent} returned empty response" >&2 + _adapter_fail "CLI_ERROR: ${agent} returned an empty response. ${ADAPTER_FALLBACK}" + return 1 fi update_agent_stats "$state_dir" "$agent" "true" record_usage "$agent" "$chars_in" "${#stdout}" diff --git a/plugin/lib/gemini-wrapper.sh b/plugin/lib/gemini-wrapper.sh index f772c02..cc96fdd 100755 --- a/plugin/lib/gemini-wrapper.sh +++ b/plugin/lib/gemini-wrapper.sh @@ -118,7 +118,7 @@ invoke_gemini_with_files() { # expand backslash escapes INSIDE file contents, corrupting code) local nl=$'\n' local file_content="" - local token path f manifest="$files_arg" + local token path f manifest="$files_arg" resolved_parent local project_root="${CLAUDE_PROJECT_DIR:-.}" local max_bytes="${DEVSQUAD_CONTEXT_MAX_BYTES:-1048576}" local max_file_bytes="${DEVSQUAD_CONTEXT_MAX_FILE_BYTES:-262144}" @@ -153,6 +153,13 @@ invoke_gemini_with_files() { while IFS= read -r -d '' f; do matched="true" [[ -L "$project_root/$f" ]] && { echo "CONTEXT_OMITTED: symlink input is not followed: ${f}" >&2; continue; } + resolved_parent=$(cd "$(dirname "$project_root/$f")" 2>/dev/null && pwd -P) || { + echo "CONTEXT_OMITTED: file parent is unavailable: ${f}" >&2; continue; + } + case "${resolved_parent}/" in + "${project_root}/"*) ;; + *) echo "CONTEXT_OMITTED: symlink ancestor escapes project scope: ${f}" >&2; continue ;; + esac case "/$f" in */.devsquad/*|*/.env|*/.env.*|*/credentials.json|*.pem|*.key) echo "CONTEXT_OMITTED: sensitive or runtime path excluded: ${f}" >&2; continue ;; @@ -177,7 +184,7 @@ invoke_gemini_with_files() { fi file_content+="=== ${f} ===${nl}$(cat "$project_root/$f")${nl}${nl}" used_bytes=$(( used_bytes + file_bytes + header_bytes )) - done < <(git -C "$project_root" ls-files -z -- "$path" 2>/dev/null) + done < <(git -C "$project_root" ls-files -z -- ":(literal)$path" 2>/dev/null) if [[ "$matched" == "false" ]]; then echo "CONTEXT_OMITTED: no tracked files in scope: ${path}" >&2 fi diff --git a/test/core/fakes/codex_app_server.py b/test/core/fakes/codex_app_server.py new file mode 100644 index 0000000..3a61815 --- /dev/null +++ b/test/core/fakes/codex_app_server.py @@ -0,0 +1,20 @@ +#!/usr/bin/env python3 +import json, sys, time +for line in sys.stdin: + request = json.loads(line) + method = request.get("method") + if method == "initialize": + print(json.dumps({"id":request["id"],"result":{"serverInfo":{"name":"fake","version":"1"}}}), flush=True) + elif method == "model/list": + cursor = request.get("params",{}).get("cursor") + result = {"data":[{"id":"gpt-fake","supportedReasoningEfforts":[{"reasoningEffort":"low","description":"fixture"}]}],"nextCursor":"two"} if cursor is None else {"data":[],"nextCursor":None} + print(json.dumps({"method":"account/updated","params":{"reason":"fixture"}}), flush=True) + print(json.dumps({"id":request["id"],"result":result}), flush=True) + elif method == "turn/start": + print(json.dumps({"id":request["id"],"result":{"turn":{"id":"turn-1"}}}), flush=True) + time.sleep(0.02) + print(json.dumps({"method":"turn/started","params":{"threadId":request["params"]["threadId"],"turn":{"id":"turn-1","status":"inProgress"}}}), flush=True) + time.sleep(0.02) + print(json.dumps({"method":"item/agentMessage/delta","params":{"threadId":request["params"]["threadId"],"turnId":"turn-1","delta":"ok"}}), flush=True) + time.sleep(0.02) + print(json.dumps({"method":"turn/completed","params":{"threadId":request["params"]["threadId"],"turn":{"id":"turn-1","status":"completed"}}}), flush=True) diff --git a/test/core/test_m1.py b/test/core/test_m1.py index 93bc16f..0a502b0 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -4,6 +4,7 @@ import os import stat import tempfile +import subprocess import unittest from pathlib import Path from unittest.mock import patch @@ -12,11 +13,11 @@ import sys sys.path.insert(0, str(CORE / "src")) -from devsquad.adapters import AdapterManifest, classify_cli, prepare_cli, prepare_native_codex +from devsquad.adapters import AdapterManifest, classify_cli, prepare_cli, prepare_native_codex, prepare_native_codex_from_catalog from devsquad.catalog import update_last_good -from devsquad.codex_protocol import NativeTurnState, model_list_request, parse_model_page +from devsquad.codex_protocol import JsonLinePeer, NativeTurnState, collect_model_pages, discover_models, initialize_request, model_list_request, parse_model_page, receive_response, review_start_request, thread_start_request, turn_interrupt_request, turn_start_request from devsquad.contracts import ContractError, validate_launch_payload -from devsquad.validation import validate_task +from devsquad.validation import validate_policy, validate_profile, validate_task class M1ContractsTest(unittest.TestCase): @@ -54,6 +55,7 @@ def test_classifier_does_not_accept_empty_denied_or_malformed_exit_zero(self): self.assertEqual(classify_cli(spec, returncode=0, stdout="", stderr="").execution_status, "malformed") self.assertEqual(classify_cli(spec, returncode=0, stdout='{"type":"result","is_error":true,"error":"tool denied"}', stderr="").execution_status, "denied") self.assertEqual(classify_cli(spec, returncode=0, stdout="not json", stderr="").execution_status, "malformed") + self.assertEqual(classify_cli(spec, returncode=0, stdout="42", stderr="").execution_status, "malformed") def test_auth_precedes_rate_and_acceptance_is_separate(self): temp, binary = self.fake_path("grok") @@ -81,6 +83,8 @@ def test_codex_real_jsonl_shape_requires_agent_message(self): spec = prepare_cli(self.manifest("codex"), prompt="x", cwd=temp.name, model=None, effort=None, permission="read_only", timeout_seconds=9) valid = '\n'.join([json.dumps({"type":"thread.started","thread_id":"x"}), json.dumps({"type":"item.completed","item":{"type":"agent_message","text":"done"}}), json.dumps({"type":"turn.completed","usage":{"input_tokens":1}})]) self.assertEqual(classify_cli(spec, returncode=0, stdout=valid, stderr="").execution_status, "succeeded") + partial = json.dumps({"type":"item.completed","item":{"type":"agent_message","text":"done"}}) + self.assertEqual(classify_cli(spec, returncode=0, stdout=partial, stderr="").execution_status, "malformed") startup = json.dumps({"type":"thread.started","thread_id":"x"}) + "\n" + json.dumps({"type":"turn.completed","usage":{}}) self.assertEqual(classify_cli(spec, returncode=0, stdout=startup, stderr="").execution_status, "malformed") @@ -95,6 +99,17 @@ def test_native_launch_is_preparation_only_and_version_scoped(self): with self.assertRaises(ContractError): prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli future") + def test_discovered_snapshot_feeds_native_preparation_and_rejects_drift(self): + temp, binary = self.fake_path("codex"); self.addCleanup(temp.cleanup) + manifest = self.manifest("codex") + snapshot = {"complete":True,"harness":"codex","harness_version":"codex-cli 0.135.0","models":[{"id":"gpt-test","supported_efforts":["low"]}]} + with patch.dict(os.environ, {"PATH": str(binary.parent)}): + spec = prepare_native_codex_from_catalog(manifest, snapshot, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.135.0") + self.assertEqual(spec.requested.model, "gpt-test") + drifted = dict(snapshot); drifted["harness_version"] = "codex-cli future" + with self.assertRaises(ContractError): + prepare_native_codex_from_catalog(manifest, drifted, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.135.0") + def test_launch_round_trip_is_strict(self): temp, binary = self.fake_path("grok"); self.addCleanup(temp.cleanup) with patch.dict(os.environ, {"PATH": str(binary.parent)}): @@ -111,8 +126,52 @@ def test_task_examples_validate_and_unknown_fields_fail(self): task["unknown"] = 1 with self.assertRaises(ContractError): validate_task(task) + def test_nested_task_override_is_strict(self): + root = Path(__file__).resolve().parents[2] + task = json.loads((root / "docs/plans/engineering-team/examples/issue-delivery.json").read_text()) + task["routing"]["overrides"] = {"reviewer":{"profile_id":"x", "fallback":"anything", "unknown":True}} + with self.assertRaises(ContractError): validate_task(task) + + def test_profile_and_policy_strict_fixtures(self): + profile = {"id":"p1","harness":"codex","model_family":"gpt","model_id":"gpt-test","effort":{"value":"low","transport":"native"},"required_tools":["read"],"permission_policy":"read_only","account_pool_id":"codex-sub","billing_mode":"subscription","quality_status":"proven","evidence_refs":["e1"]} + validate_profile(profile) + broken = dict(profile); broken["surprise"] = 1 + with self.assertRaises(ContractError): validate_profile(broken) + policy = {"schema_version":1,"id":"default","version":1,"roles":{"reviewer":[{"kind":"profile","id":"p1"}]},"task_classes":{},"require_different_model_for_review":True,"account_pools":{},"experiment_budget":{}} + validate_policy(policy) + policy["roles"]["reviewer"][0]["unknown"] = True + with self.assertRaises(ContractError): validate_policy(policy) + class NativeProtocolTest(unittest.TestCase): + def test_peer_partial_frame_times_out_and_buffered_second_frame_drains(self): + read_fd, write_fd = os.pipe() + reader = os.fdopen(read_fd, "r"); writer = os.fdopen(write_fd, "w") + self.addCleanup(reader.close); self.addCleanup(writer.close) + peer = JsonLinePeer(reader, writer) + os.write(write_fd, b'{') + started = __import__('time').monotonic() + with self.assertRaises(TimeoutError): peer.receive(0.05) + self.assertLess(__import__('time').monotonic() - started, 0.2) + os.write(write_fd, b'}\n{"n":2}\n') + self.assertEqual(peer.receive(0.1), {}) + self.assertEqual(peer.receive(0.1), {"n":2}) + def test_fake_app_server_protocol_conformance(self): + fake = Path(__file__).parent / "fakes/codex_app_server.py" + process = subprocess.Popen([sys.executable, str(fake)], stdin=subprocess.PIPE, stdout=subprocess.PIPE, text=True) + def cleanup(): + if process.poll() is None: process.terminate() + process.wait(timeout=2) + process.stdin.close(); process.stdout.close() + self.addCleanup(cleanup) + peer = JsonLinePeer(process.stdout, process.stdin) + peer.send(initialize_request(1)); self.assertIn("result", receive_response(peer, 1, timeout_seconds=2)) + self.assertEqual([m["id"] for m in discover_models(peer, first_request_id=2, timeout_seconds=2)], ["gpt-fake"]) + peer.send(turn_start_request(4, thread_id="thread-1", prompt="p", model="gpt-fake", effort="low", cwd="/tmp", permission="read_only")) + response = receive_response(peer, 4, timeout_seconds=2); self.assertEqual(response["result"]["turn"]["id"], "turn-1") + state = NativeTurnState(thread_id="thread-1", turn_id="turn-1") + for _ in range(3): state.consume(peer.receive(2)) + self.assertTrue(state.terminal); self.assertEqual("".join(state.output), "ok") def test_paginated_model_request_and_response(self): self.assertEqual(model_list_request(2, "next")["params"]["cursor"], "next") models, cursor = parse_model_page({"result": {"data": [{"id": "gpt-x"}], "nextCursor": "c2"}}) @@ -120,18 +179,39 @@ def test_paginated_model_request_and_response(self): self.assertEqual(cursor, "c2") def test_start_and_interrupt_ack_are_not_terminal(self): - state = NativeTurnState() - state.consume({"method": "turn/started", "params": {"turn": {"id": "t1"}}}) - state.consume({"method": "turn/interrupt/completed", "params": {}}) + state = NativeTurnState(thread_id="th1") + state.consume({"method": "turn/started", "params": {"threadId": "th1", "turn": {"id": "t1", "status":"inProgress"}}}) + state.acknowledge_interrupt({"id": 4, "result": {}}) self.assertFalse(state.terminal) self.assertTrue(state.interrupted_acknowledged) - state.consume({"method": "turn/interrupted", "params": {}}) + state.consume({"method": "turn/completed", "params": {"threadId":"th1", "turn":{"id":"t1", "status":"interrupted"}}}) self.assertTrue(state.terminal) + def test_unrelated_turn_cannot_complete_ours_and_disconnect_is_visible(self): + state = NativeTurnState(thread_id="th1", turn_id="ours") + state.consume({"method":"turn/completed", "params":{"threadId":"th1", "turn":{"id":"other", "status":"completed"}}}) + self.assertFalse(state.terminal) + state.disconnected(); self.assertEqual(state.terminal_status, "transport_disconnected") + + def test_typed_native_requests_match_installed_contract(self): + thread = thread_start_request(1, cwd="/tmp/repo", model="gpt-test", permission="read_only") + self.assertEqual(thread["params"]["sandbox"], "read-only") + turn = turn_start_request(2, thread_id="th", prompt="p", model="gpt-test", effort="low", cwd="/tmp/repo", permission="read_only", output_schema={"type":"object"}) + self.assertEqual(turn["params"]["sandboxPolicy"], {"type":"readOnly", "networkAccess":False}) + self.assertEqual(turn["params"]["outputSchema"]["type"], "object") + self.assertEqual(turn_interrupt_request(3, thread_id="th", turn_id="tu")["params"]["turnId"], "tu") + self.assertEqual(review_start_request(4, thread_id="th", target={"type":"commit", "sha":"abc"})["method"], "review/start") + def test_disconnect_or_malformed_page_is_visible(self): with self.assertRaises(ContractError): parse_model_page({"result": {"nextCursor": "never"}}) + def test_complete_pagination_empty_page_and_repeated_cursor(self): + pages = {None:{"result":{"data":[{"id":"a"}],"nextCursor":"c"}}, "c":{"result":{"data":[],"nextCursor":None}}} + self.assertEqual([m["id"] for m in collect_model_pages(lambda c: pages[c])], ["a"]) + with self.assertRaises(ContractError): + collect_model_pages(lambda c: {"result":{"data":[],"nextCursor":"same"}}) + class CatalogTest(unittest.TestCase): def test_incomplete_refresh_retains_last_good(self): diff --git a/test/test_m1_legacy.sh b/test/test_m1_legacy.sh index 4165cd5..9d1db10 100755 --- a/test/test_m1_legacy.sh +++ b/test/test_m1_legacy.sh @@ -8,7 +8,7 @@ bad() { FAIL=$((FAIL + 1)); echo " FAIL: $1"; } T=$(mktemp -d) trap 'rm -rf "$T"' EXIT -mkdir -p "$T/bin" "$T/project/.devsquad" +mkdir -p "$T/bin" "$T/home" "$T/project/.devsquad" cat > "$T/bin/codex" <<'EOF' #!/usr/bin/env bash case "${FAKE_MODE:-fast}" in @@ -17,17 +17,40 @@ case "${FAKE_MODE:-fast}" in sh -c 'trap "" TERM; echo $$ > "$DESC_PID_FILE"; while :; do sleep 1; done' & wait ;; + root_ignore) + trap '' TERM + printf '%s\n' "$$" > "$ROOT_PID_FILE" + : > "$ROOT_READY_FILE" + while :; do sleep 1; done + ;; esac EOF chmod +x "$T/bin/codex" +# Record the portable watchdog's timer PID so the success path can prove it +# did not orphan the sleep process. Other sleep durations use the real binary. +cat > "$T/bin/sleep" <<'EOF' +#!/usr/bin/env bash +if [[ "$1" == "2" && -n "${WATCHDOG_SLEEP_PID_FILE:-}" ]]; then + printf '%s\n' "$$" > "$WATCHDOG_SLEEP_PID_FILE" +fi +exec /bin/sleep "$@" +EOF +chmod +x "$T/bin/sleep" + # Force the portable path even on systems with timeout/gtimeout installed. -start=$(date +%s) -PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=fast \ - bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; invoke_codex hello 10 2' _ "$ROOT" > "$T/out" 2> "$T/err" -fast_elapsed=$(( $(date +%s) - start )) +WATCHDOG_SLEEP_PID_FILE="$T/watchdog-sleep.pid"; export WATCHDOG_SLEEP_PID_FILE +FAST_ELAPSED_FILE="$T/fast-elapsed"; export FAST_ELAPSED_FILE +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=fast \ + bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; s=$(perl -MTime::HiRes=time -e '\''printf "%.6f", time'\''); invoke_codex hello 10 2; f=$(perl -MTime::HiRes=time -e '\''printf "%.6f", time'\''); awk -v s="$s" -v f="$f" '\''BEGIN { printf "%.3f", f-s }'\'' > "$FAST_ELAPSED_FILE"' _ "$ROOT" > "$T/out" 2> "$T/err" +fast_elapsed=$(cat "$FAST_ELAPSED_FILE") [[ "$(cat "$T/out")" == "ok" ]] && ok || bad "portable watchdog output" -[[ "$fast_elapsed" -lt 2 ]] && ok || bad "portable fast call waited ${fast_elapsed}s for watchdog" +awk -v e="$fast_elapsed" 'BEGIN { exit !(e < 1.0) }' && ok || bad "portable fast call took ${fast_elapsed}s" +if [[ -s "$WATCHDOG_SLEEP_PID_FILE" ]] && kill -0 "$(cat "$WATCHDOG_SLEEP_PID_FILE")" 2>/dev/null; then + bad "portable success left watchdog sleep alive" +else + ok +fi # Model lookup remains optional when jq is absent from PATH. mkdir -p "$T/nojq" @@ -38,7 +61,7 @@ PATH="$T/nojq" CLAUDE_PROJECT_DIR="$T/project" /bin/bash -c \ DESC_PID_FILE="$T/desc.pid"; export DESC_PID_FILE start=$(date +%s) -PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=tree \ +PATH="$T/bin:/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=tree \ bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; invoke_codex hello 10 1' _ "$ROOT" > "$T/tree.out" 2> "$T/tree.err" || rc=$? elapsed=$(( $(date +%s) - start )) [[ "${rc:-0}" -eq 1 ]] && grep -q '^TIMEOUT:' "$T/tree.err" && ok || bad "portable timeout classification" @@ -49,6 +72,25 @@ else ok fi +# A root process that ignores TERM must still be KILLed by the watchdog itself; +# escalation after wait would deadlock forever. The readiness marker proves the +# fake entered its signal-resistant loop before the deadline. +ROOT_PID_FILE="$T/root.pid" ROOT_READY_FILE="$T/root.ready" +export ROOT_PID_FILE ROOT_READY_FILE +start=$(perl -MTime::HiRes=time -e 'printf "%.6f", time') +HOME="$T/home" PATH="/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/codex" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=root_ignore \ + /bin/bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; invoke_codex hello 10 3' _ "$ROOT" > "$T/root.out" 2> "$T/root.err" || root_rc=$? +finish=$(perl -MTime::HiRes=time -e 'printf "%.6f", time') +root_elapsed=$(awk -v s="$start" -v f="$finish" 'BEGIN { printf "%.3f", f-s }') +[[ -f "$ROOT_READY_FILE" ]] && ok || bad "TERM-ignoring root never reached readiness" +[[ "${root_rc:-0}" -eq 1 ]] && grep -q '^TIMEOUT:' "$T/root.err" && ok || bad "TERM-ignoring root timeout classification" +awk -v e="$root_elapsed" 'BEGIN { exit !(e >= 2.8 && e < 5.0) }' && ok || bad "TERM-ignoring root took ${root_elapsed}s" +if [[ -s "$ROOT_PID_FILE" ]] && kill -0 "$(cat "$ROOT_PID_FILE")" 2>/dev/null; then + bad "portable timeout left TERM-ignoring root alive" +else + ok +fi + # Newline manifests preserve spaces and enumerate TSX. mkdir -p "$T/project/ui/My Folder" printf 'export const Card = 1;\n' > "$T/project/ui/My Folder/Card.tsx" @@ -80,5 +122,18 @@ grep -q 'file exceeds 4 byte limit: large.ts' "$T/omitted.err" && ok || bad "byt grep -q 'symlink input is not followed: escape.ts' "$T/omitted.err" && ok || bad "symlink omission not reported" grep -q 'path escapes project scope: ../outside' "$T/omitted.err" && ok || bad "path escape not reported" +# Replacing a tracked directory with a symlink must not let a tracked path read +# bytes outside the project through a symlink ancestor. +mkdir -p "$T/project/safe" "$T/outside" +printf 'inside\n' > "$T/project/safe/code.ts" +git -C "$T/project" add safe/code.ts +mv "$T/project/safe" "$T/project/safe.real" +printf 'OUTSIDE_MARKER\n' > "$T/outside/code.ts" +ln -s "$T/outside" "$T/project/safe" +PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" \ + bash -c 'source "$1/plugin/lib/gemini-wrapper.sh"; _adapter_invoke() { cat "$ADAPTER_STDIN_FILE"; }; invoke_gemini_with_files "@safe/code.ts" inspect 10 2' _ "$ROOT" > "$T/ancestor.out" 2> "$T/ancestor.err" +grep -q 'OUTSIDE_MARKER' "$T/ancestor.out" && bad "symlink ancestor leaked outside content" || ok +grep -q 'symlink ancestor escapes project scope: safe/code.ts' "$T/ancestor.err" && ok || bad "symlink ancestor omission not reported" + echo " m1_legacy: ${PASS} passed, ${FAIL} failed" [[ "$FAIL" -eq 0 ]] diff --git a/test/test_wrapper_contract.sh b/test/test_wrapper_contract.sh index 5c4d56f..ce591e8 100644 --- a/test/test_wrapper_contract.sh +++ b/test/test_wrapper_contract.sh @@ -27,6 +27,7 @@ case "${FAKE_MODE:-success}" in auth) echo "401 unauthorized request" >&2; exit 1 ;; migrate) echo "please migrate to the new suite: IneligibleTierError while authenticating" >&2; exit 1 ;; banner) echo "Signing in with Grok..." ;; + empty) : ;; esac FAKESH chmod +x "$FAKE/$bin" @@ -76,6 +77,12 @@ for spec in "gemini-wrapper.sh:invoke_gemini:gemini" "codex-wrapper.sh:invoke_co [ -f "$TDIR/.devsquad/usage/$agent.json" ] && ok || bad "$agent failure usage record" done +# Exit zero without a usable response is a contract failure, not success. +run_case codex-wrapper.sh invoke_codex empty +[ "$EC" -ne 0 ] && ok || bad "codex empty exit-zero treated as success" +printf '%s' "$ERR_TXT" | grep -q '^CLI_ERROR:' && ok || bad "codex empty exit-zero prefix" +[ -f "$TDIR/.devsquad/usage/codex.json" ] && ok || bad "codex empty exit-zero usage record" + # grok-specific: unauthenticated CLI exits 0 with a sign-in banner — the # wrapper must classify that as AUTH_ERROR, not success run_case grok-wrapper.sh invoke_grok banner From a67ab58c4f5d2d298b7b7c6ba1e2fa3fc3732b9d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 01:00:08 +0530 Subject: [PATCH 011/197] fix: satisfy M1 protocol and validation gates --- .../core/adapters/classification-policy.conf | 4 + plugin/core/pyproject.toml | 1 + .../schemas/execution-identity.schema.json | 8 +- plugin/core/schemas/launch-spec.schema.json | 4 +- .../schemas/normalized-result.schema.json | 2 +- plugin/core/schemas/policy.schema.json | 4 +- plugin/core/schemas/profile.schema.json | 4 +- plugin/core/schemas/task.schema.json | 4 +- plugin/core/src/devsquad/adapters.py | 27 +++- plugin/core/src/devsquad/codex_protocol.py | 4 + plugin/core/src/devsquad/contracts.py | 52 ++++++- plugin/core/src/devsquad/validation.py | 58 ++++++-- plugin/lib/adapter.sh | 42 +++--- test/core/fakes/codex_app_server.py | 7 + test/core/test_m1.py | 7 +- test/core/test_m1_gate_review.py | 140 ++++++++++++++++++ test/core/test_validation.py | 69 +++++++++ test/test_m1_legacy.sh | 13 +- 18 files changed, 388 insertions(+), 62 deletions(-) create mode 100644 plugin/core/adapters/classification-policy.conf create mode 100644 test/core/test_m1_gate_review.py create mode 100644 test/core/test_validation.py diff --git a/plugin/core/adapters/classification-policy.conf b/plugin/core/adapters/classification-policy.conf new file mode 100644 index 0000000..825277b --- /dev/null +++ b/plugin/core/adapters/classification-policy.conf @@ -0,0 +1,4 @@ +# Shared legacy/new bridge taxonomy. Extended orchestration states live above it. +DEVSQUAD_AUTH_ERROR_PATTERN='auth|unauthorized|ineligible|(^|[^0-9])401([^0-9]|$)|(^|[^0-9])403([^0-9]|$)' +DEVSQUAD_RATE_LIMIT_PATTERN='rate.?limit|quota|resource.?exhausted|too many requests|(^|[^0-9])429([^0-9]|$)' +DEVSQUAD_DENIED_PATTERN='permission denied|tool (use )?denied|not allowed' diff --git a/plugin/core/pyproject.toml b/plugin/core/pyproject.toml index ef8f93b..22b82fc 100644 --- a/plugin/core/pyproject.toml +++ b/plugin/core/pyproject.toml @@ -21,5 +21,6 @@ where = ["src"] "share/devsquad/adapters/codex" = ["adapters/codex/adapter.json"] "share/devsquad/adapters/antigravity" = ["adapters/antigravity/adapter.json"] "share/devsquad/adapters/grok" = ["adapters/grok/adapter.json"] +"share/devsquad/adapters" = ["adapters/classification-policy.conf"] "share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/task.schema.json"] "share/devsquad/profiles" = ["profiles/templates.json"] diff --git a/plugin/core/schemas/execution-identity.schema.json b/plugin/core/schemas/execution-identity.schema.json index 5b248ed..80cab06 100644 --- a/plugin/core/schemas/execution-identity.schema.json +++ b/plugin/core/schemas/execution-identity.schema.json @@ -4,10 +4,10 @@ "type": "object", "additionalProperties": false, "required": ["harness", "harness_version", "model_provider", "model_family", "model", "effort", "tools", "permissions", "account_pool", "verification"], "properties": { - "harness": {"type": "string", "minLength": 1}, "harness_version": {"type": ["string", "null"]}, - "model_provider": {"type": ["string", "null"]}, "model_family": {"type": ["string", "null"]}, "model": {"type": ["string", "null"]}, - "effort": {"type": ["string", "null"]}, "tools": {"type": "array", "items": {"type": "string"}, "uniqueItems": true}, - "permissions": {"enum": ["read_only", "workspace_write"]}, "account_pool": {"type": ["string", "null"]}, + "harness": {"type": "string", "minLength": 1}, "harness_version": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, + "model_provider": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, "model_family": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, "model": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, + "effort": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, "tools": {"type": "array", "items": {"type": "string", "minLength":1}, "uniqueItems": true}, + "permissions": {"enum": ["read_only", "workspace_write"]}, "account_pool": {"oneOf":[{"type":"string","minLength":1},{"type":"null"}]}, "verification": {"enum": ["verified", "unverified", "unavailable", "unknown"]} } } diff --git a/plugin/core/schemas/launch-spec.schema.json b/plugin/core/schemas/launch-spec.schema.json index aa6d282..481d795 100644 --- a/plugin/core/schemas/launch-spec.schema.json +++ b/plugin/core/schemas/launch-spec.schema.json @@ -10,9 +10,9 @@ "transport": {"enum": ["cli_exec", "native_protocol"]}, "argv": {"type": "array", "minItems": 1, "items": {"type": "string", "minLength": 1}}, "cwd": {"type": "string", "minLength": 1}, - "stdin_path": {"type": ["string", "null"]}, + "stdin_path": {"oneOf": [{"type": "string", "minLength": 1}, {"type": "null"}]}, "timeout_seconds": {"type": "integer", "minimum": 1}, "requested": {"$ref": "execution-identity.schema.json"}, - "environment": {"type": "object", "additionalProperties": {"type": "string"}} + "environment": {"type": "object", "additionalProperties": false, "properties": {"DEVSQUAD_WORKER":{"type":"string"},"DEVSQUAD_RUN_ID":{"type":"string"},"DEVSQUAD_ATTEMPT_ID":{"type":"string"},"DEVSQUAD_DELEGATION_DEPTH":{"type":"string"}}} } } diff --git a/plugin/core/schemas/normalized-result.schema.json b/plugin/core/schemas/normalized-result.schema.json index e7c4f71..ae41124 100644 --- a/plugin/core/schemas/normalized-result.schema.json +++ b/plugin/core/schemas/normalized-result.schema.json @@ -13,7 +13,7 @@ "acceptance_status": {"enum": ["pending", "accepted", "rejected", "not_evaluated"]}, "requested": {"$ref": "execution-identity.schema.json"}, "observed": {"oneOf": [{"$ref": "execution-identity.schema.json"}, {"type": "null"}]}, - "native_ids": {"type": "object", "additionalProperties": {"type": "string"}}, + "native_ids": {"type": "object", "propertyNames":{"minLength":1}, "additionalProperties": {"type": "string", "minLength":1}}, "events": {"type": "array", "items": {"type": "object"}} } } diff --git a/plugin/core/schemas/policy.schema.json b/plugin/core/schemas/policy.schema.json index 3e03ae4..aaba821 100644 --- a/plugin/core/schemas/policy.schema.json +++ b/plugin/core/schemas/policy.schema.json @@ -5,8 +5,8 @@ "properties":{ "schema_version":{"const":1},"id":{"type":"string","minLength":1},"version":{"type":"integer","minimum":1}, "roles":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"type":"array","minItems":1,"items":{"$ref":"#/$defs/candidate"}}}, - "task_classes":{"type":"object"},"require_different_model_for_review":{"type":"boolean"},"prefer_different_harness_for_review":{"type":"boolean"}, - "account_pools":{"type":"object"},"experiment_budget":{"type":"object"} + "task_classes":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"enum":["unvalidated","trial","proven","suspended"]}},"require_different_model_for_review":{"type":"boolean"},"prefer_different_harness_for_review":{"type":"boolean"}, + "account_pools":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"type":"object"}},"experiment_budget":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"type":"integer","minimum":0}} }, "$defs":{"candidate":{"type":"object","additionalProperties":false,"required":["kind","id"],"properties":{"kind":{"enum":["profile","alias"]},"id":{"type":"string","minLength":1}}}} } diff --git a/plugin/core/schemas/profile.schema.json b/plugin/core/schemas/profile.schema.json index 2c874a0..4f12512 100644 --- a/plugin/core/schemas/profile.schema.json +++ b/plugin/core/schemas/profile.schema.json @@ -10,11 +10,11 @@ "model_family": {"type": "string", "minLength": 1}, "model_id": {"type": "string", "minLength": 1}, "effort": {"type": "object", "additionalProperties": false, "required": ["value", "transport"], "properties": {"value": {"type": ["string", "null"]}, "transport": {"enum": ["native", "model_variant", "provider_default"]}}}, - "required_tools": {"type": "array", "items": {"type": "string"}, "uniqueItems": true}, + "required_tools": {"type": "array", "items": {"type": "string", "minLength":1}, "uniqueItems": true}, "permission_policy": {"enum": ["read_only", "workspace_write"]}, "account_pool_id": {"type": "string", "minLength": 1}, "billing_mode": {"enum": ["subscription", "paid_api"]}, "quality_status": {"enum": ["unvalidated", "trial", "proven", "suspended"]}, - "evidence_refs": {"type": "array", "items": {"type": "string"}, "uniqueItems": true} + "evidence_refs": {"type": "array", "items": {"type": "string", "minLength":1}, "uniqueItems": true} } } diff --git a/plugin/core/schemas/task.schema.json b/plugin/core/schemas/task.schema.json index d452a8b..96c9830 100644 --- a/plugin/core/schemas/task.schema.json +++ b/plugin/core/schemas/task.schema.json @@ -14,12 +14,12 @@ "project": {"type":"object","additionalProperties":false,"required":["repo_path","base_ref","target_ref"],"properties":{"repo_path":{"type":"string","pattern":"^/"},"base_ref":{"type":"string","minLength":1},"target_ref":{"type":"string","minLength":1}}}, "acceptance": {"type":"object","additionalProperties":false,"required":["id","description","evidence_kind"],"properties":{"id":{"type":"string","minLength":1},"description":{"type":"string","minLength":1},"evidence_kind":{"enum":["review","check","artifact","host"]}}}, "check": {"type":"object","additionalProperties":false,"required":["id","argv","cwd","timeout_seconds","required_to_pass"],"properties":{"id":{"type":"string","minLength":1},"argv":{"type":"array","minItems":1,"items":{"type":"string","minLength":1}},"cwd":{"type":"string"},"timeout_seconds":{"type":"integer","minimum":1},"required_to_pass":{"type":"boolean"}}}, - "scope": {"type":"object","additionalProperties":false,"required":["read_paths","write_paths"],"properties":{"read_paths":{"type":"array","items":{"type":"string"}},"write_paths":{"type":"array","items":{"type":"string"}}}}, + "scope": {"type":"object","additionalProperties":false,"required":["read_paths","write_paths"],"properties":{"read_paths":{"type":"array","uniqueItems":true,"items":{"type":"string","minLength":1}},"write_paths":{"type":"array","uniqueItems":true,"items":{"type":"string","minLength":1}}}}, "lead": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["host","headless"]}}}, "override": {"type":"object","additionalProperties":false,"required":["profile_id"],"properties":{"profile_id":{"type":"string","minLength":1},"fallback":{"enum":["none","policy"]}}}, "routing": {"type":"object","additionalProperties":false,"required":["profiles_file","policy_file"],"properties":{"profiles_file":{"type":"string","minLength":1},"policy_file":{"type":"string","minLength":1},"overrides":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"$ref":"#/$defs/override"}}}}, "budget": {"type":"object","additionalProperties":false,"required":["wall_seconds","max_worker_invocations","max_revisions","max_fallbacks_per_step"],"properties":{"wall_seconds":{"type":"integer","minimum":1},"max_worker_invocations":{"type":"integer","minimum":1},"max_revisions":{"type":"integer","minimum":0},"max_fallbacks_per_step":{"type":"integer","minimum":0}}}, - "origin": {"type":"object","additionalProperties":false,"required":["surface"],"properties":{"surface":{"type":"string","minLength":1},"session_ref":{"type":"string"}}}, + "origin": {"type":"object","additionalProperties":false,"required":["surface"],"properties":{"surface":{"type":"string","minLength":1},"session_ref":{"type":"string","minLength":1}}}, "review": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["standard","adversarial"]},"focus":{"type":"string","minLength":1}}} } } diff --git a/plugin/core/src/devsquad/adapters.py b/plugin/core/src/devsquad/adapters.py index c493a32..52166a7 100644 --- a/plugin/core/src/devsquad/adapters.py +++ b/plugin/core/src/devsquad/adapters.py @@ -7,6 +7,8 @@ import re import shutil import subprocess +import shlex +import sys from dataclasses import dataclass from pathlib import Path from typing import Any @@ -14,10 +16,20 @@ from .contracts import ContractError, ExecutionIdentity, LaunchSpec, NormalizedResult, ProfileUnsupported, SCHEMA_VERSION from .catalog import verified_efforts -ERROR_PATTERNS = ( - ("AUTH_ERROR", re.compile(r"auth|unauthorized|ineligible|\b401\b|\b403\b", re.I)), - ("RATE_LIMITED", re.compile(r"rate.?limit|quota|resource.?exhausted|too many requests|\b429\b", re.I)), -) +def _classification_policy() -> dict[str, str]: + source = Path(__file__).resolve().parents[2] / "adapters" / "classification-policy.conf" + if not source.exists(): + source = Path(sys.prefix) / "share" / "devsquad" / "adapters" / "classification-policy.conf" + values = {} + for line in source.read_text().splitlines(): + if line and not line.startswith("#"): + key, raw = line.split("=", 1); values[key] = shlex.split(raw)[0] + return values + + +_POLICY = _classification_policy() +ERROR_PATTERNS = (("AUTH_ERROR", re.compile(_POLICY["DEVSQUAD_AUTH_ERROR_PATTERN"], re.I)), ("RATE_LIMITED", re.compile(_POLICY["DEVSQUAD_RATE_LIMIT_PATTERN"], re.I))) +DENIED_PATTERN = re.compile(_POLICY["DEVSQUAD_DENIED_PATTERN"], re.I) @dataclass(frozen=True) @@ -120,7 +132,8 @@ def prepare_native_codex(manifest: AdapterManifest, *, cwd: str, model: str, eff raise ProfileUnsupported(f"unsupported or unverified effort {effort!r} for codex model {model!r}") _permission_args(manifest, permission) requested = ExecutionIdentity("codex", harness_version_value, "openai", None, model, effort, (), permission, None, "verified") - return LaunchSpec(SCHEMA_VERSION, "codex", "native_protocol", (binary, "app-server", "--listen", "stdio://"), str(Path(cwd).resolve()), None, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) + argv = (binary, "-c", f'model="{model}"', "-c", f'model_reasoning_effort="{effort}"', "app-server", "--listen", "stdio://") + return LaunchSpec(SCHEMA_VERSION, "codex", "native_protocol", argv, str(Path(cwd).resolve()), None, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) def prepare_native_codex_from_catalog(manifest: AdapterManifest, snapshot: dict[str, Any], *, cwd: str, model: str, effort: str, permission: str, timeout_seconds: int, harness_version_value: str) -> LaunchSpec: @@ -149,6 +162,8 @@ def _provider_records(adapter: str, stdout: str) -> tuple[list[dict[str, Any]], terminal = True if adapter == "codex" and kind == "item.completed": native_item = item.get("item") or {} + if not isinstance(native_item, dict): + raise json.JSONDecodeError("item.completed item is not an object", stdout, 0) if native_item.get("type") in {"agent_message", "agentMessage"}: payload = native_item.get("text") or native_item.get("content") deliverable = isinstance(payload, str) and bool(payload.strip()) @@ -175,7 +190,7 @@ def classify_cli(spec: LaunchSpec, *, returncode: int, stdout: str, stderr: str, _, deliverable, terminal, provider_error = _provider_records(spec.adapter, stdout) if provider_error: code = next((candidate for candidate, pattern in ERROR_PATTERNS if pattern.search(provider_error)), "CLI_ERROR") - status = "denied" if re.search(r"permission denied|tool (?:use )?denied|not allowed", provider_error, re.I) else "failed" + status = "denied" if DENIED_PATTERN.search(provider_error) else "failed" elif not deliverable or not terminal: status, code = "malformed", "CLI_ERROR" except json.JSONDecodeError: diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index d975f60..6439f34 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -66,6 +66,10 @@ def initialize_request(request_id: int = 1) -> dict[str, Any]: return request(request_id, "initialize", {"clientInfo": {"name": "devsquad", "version": "0.1.0"}, "capabilities": {"experimentalApi": True}}) +def initialized_notification() -> dict[str, Any]: + return {"method": "initialized", "params": {}} + + def thread_start_request(request_id: int, *, cwd: str, model: str, permission: str) -> dict[str, Any]: sandbox = {"read_only": "read-only", "workspace_write": "workspace-write"}.get(permission) if sandbox is None: diff --git a/plugin/core/src/devsquad/contracts.py b/plugin/core/src/devsquad/contracts.py index dd8f584..fcfece4 100644 --- a/plugin/core/src/devsquad/contracts.py +++ b/plugin/core/src/devsquad/contracts.py @@ -37,12 +37,18 @@ class ExecutionIdentity: def __post_init__(self) -> None: if not isinstance(self.harness, str) or not self.harness: raise ContractError("identity harness must be non-empty") - if self.permissions not in {"read_only", "workspace_write"}: + if not isinstance(self.permissions, str) or self.permissions not in {"read_only", "workspace_write"}: raise ContractError("identity permission is invalid") - if self.verification not in {"verified", "unverified", "unavailable", "unknown"}: + if not isinstance(self.verification, str) or self.verification not in {"verified", "unverified", "unavailable", "unknown"}: raise ContractError("identity verification is invalid") if not isinstance(self.tools, tuple) or not all(isinstance(v, str) for v in self.tools): raise ContractError("identity tools must be a string tuple") + if len(set(self.tools)) != len(self.tools) or any(not v for v in self.tools): + raise ContractError("identity tools must contain unique non-empty strings") + for name in ("harness_version", "model_provider", "model_family", "model", "effort", "account_pool"): + value = getattr(self, name) + if value is not None and (not isinstance(value, str) or not value): + raise ContractError(f"identity {name} must be a non-empty string or null") @dataclass(frozen=True) @@ -60,16 +66,22 @@ class LaunchSpec: environment: dict[str, str] = field(default_factory=dict) def __post_init__(self) -> None: - if self.schema_version != SCHEMA_VERSION: + if type(self.schema_version) is not int or self.schema_version != SCHEMA_VERSION: raise ContractError("unsupported schema_version") - if self.transport not in ("cli_exec", "native_protocol"): + if not isinstance(self.transport, str) or self.transport not in ("cli_exec", "native_protocol"): raise ContractError("unsupported transport") - if not self.argv or not all(isinstance(v, str) and v for v in self.argv): + if not isinstance(self.adapter, str) or not self.adapter: + raise ContractError("adapter must be non-empty") + if not isinstance(self.argv, tuple) or not self.argv or not all(isinstance(v, str) and v for v in self.argv): raise ContractError("argv must be a non-empty string array") - if self.timeout_seconds <= 0: + if type(self.timeout_seconds) is not int or self.timeout_seconds <= 0: raise ContractError("timeout_seconds must be positive") - if not Path(self.cwd).is_absolute(): + if not isinstance(self.cwd, str) or not self.cwd or not Path(self.cwd).is_absolute(): raise ContractError("cwd must be absolute") + if self.stdin_path is not None and (not isinstance(self.stdin_path, str) or not self.stdin_path): + raise ContractError("stdin_path must be a non-empty string or null") + if not isinstance(self.requested, ExecutionIdentity): + raise ContractError("requested must be an execution identity") allowed_env = {"DEVSQUAD_WORKER", "DEVSQUAD_RUN_ID", "DEVSQUAD_ATTEMPT_ID", "DEVSQUAD_DELEGATION_DEPTH"} if not isinstance(self.environment, dict) or set(self.environment) - allowed_env or not all(isinstance(k, str) and isinstance(v, str) for k, v in self.environment.items()): raise ContractError("environment contains non-allowlisted or non-string values") @@ -95,6 +107,26 @@ class NormalizedResult: native_ids: dict[str, str] = field(default_factory=dict) events: tuple[dict[str, Any], ...] = () + def __post_init__(self) -> None: + if type(self.schema_version) is not int or self.schema_version != SCHEMA_VERSION: + raise ContractError("unsupported schema_version") + if not isinstance(self.execution_status, str) or self.execution_status not in {"succeeded", "failed", "timed_out", "interrupted", "denied", "malformed"}: + raise ContractError("invalid execution_status") + if not isinstance(self.artifact_status, str) or self.artifact_status not in {"present", "missing", "not_required", "unknown"}: + raise ContractError("invalid artifact_status") + if not isinstance(self.acceptance_status, str) or self.acceptance_status not in {"pending", "accepted", "rejected", "not_evaluated"}: + raise ContractError("invalid acceptance_status") + for name in ("error_code", "output"): + value = getattr(self, name) + if value is not None and not isinstance(value, str): + raise ContractError(f"{name} must be a string or null") + if not isinstance(self.requested, ExecutionIdentity) or (self.observed is not None and not isinstance(self.observed, ExecutionIdentity)): + raise ContractError("requested/observed identity is invalid") + if not isinstance(self.native_ids, dict) or not all(isinstance(k, str) and k and isinstance(v, str) and v for k, v in self.native_ids.items()): + raise ContractError("native_ids must contain non-empty string pairs") + if not isinstance(self.events, tuple) or not all(isinstance(v, dict) for v in self.events): + raise ContractError("events must be an object tuple") + def to_dict(self) -> dict[str, Any]: return asdict(self) @@ -110,7 +142,13 @@ def error_payload(code: str, message: str, *, retryable: bool = False, details: def validate_launch_payload(value: dict[str, Any]) -> None: + if not isinstance(value, dict): + raise ContractError("LaunchSpec must be an object") expected = {"schema_version", "adapter", "transport", "argv", "cwd", "stdin_path", "timeout_seconds", "requested", "environment"} if set(value) != expected: raise ContractError(f"LaunchSpec fields differ: {sorted(set(value) ^ expected)}") + if not isinstance(value["argv"], (list, tuple)) or isinstance(value["argv"], (str, bytes)): + raise ContractError("argv must be an array") + if not isinstance(value["requested"], dict) or not isinstance(value["requested"].get("tools"), (list, tuple)): + raise ContractError("requested identity is invalid") LaunchSpec(**{**value, "argv": tuple(value["argv"]), "requested": ExecutionIdentity(**{**value["requested"], "tools": tuple(value["requested"]["tools"])})}) diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py index 4724aff..5879b8c 100644 --- a/plugin/core/src/devsquad/validation.py +++ b/plugin/core/src/devsquad/validation.py @@ -17,6 +17,8 @@ def _exact(value: dict[str, Any], allowed: set[str], required: set[str], label: def _relative(path: str, label: str) -> None: + if not isinstance(path, str) or not path: + raise ContractError(f"{label} must be a non-empty string") p = Path(path) if p.is_absolute() or ".." in p.parts: raise ContractError(f"{label} must be repository-relative without traversal") @@ -24,11 +26,11 @@ def _relative(path: str, label: str) -> None: def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) -> None: _exact(value, TASK_FIELDS, TASK_FIELDS - {"review"}, "task") - if type(value["schema_version"]) is not int or value["schema_version"] != 1 or value["workflow"] not in {"branch-review", "issue-delivery"}: + if type(value["schema_version"]) is not int or value["schema_version"] != 1 or not isinstance(value["workflow"], str) or value["workflow"] not in {"branch-review", "issue-delivery"}: raise ContractError("unsupported task schema or workflow") project = value["project"] _exact(project, {"repo_path", "base_ref", "target_ref"}, {"repo_path", "base_ref", "target_ref"}, "project") - if not Path(project["repo_path"]).is_absolute() or (require_existing_repo and not (Path(project["repo_path"]) / ".git").exists()): + if not all(isinstance(project[k], str) and project[k] for k in ("repo_path", "base_ref", "target_ref")) or not Path(project["repo_path"]).is_absolute() or (require_existing_repo and not (Path(project["repo_path"]) / ".git").exists()): raise ContractError("project.repo_path must be an existing absolute Git repository") if not isinstance(value["goal"], str) or not value["goal"].strip() or not isinstance(value["task_class"], str) or not value["task_class"].strip(): raise ContractError("goal and task_class must be non-empty strings") @@ -36,21 +38,25 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) raise ContractError("acceptance must be non-empty") for item in value["acceptance"]: _exact(item, {"id", "description", "evidence_kind"}, {"id", "description", "evidence_kind"}, "acceptance item") - if item["evidence_kind"] not in {"review", "check", "artifact", "host"}: + if not all(isinstance(item[k], str) and item[k].strip() for k in ("id", "description")): + raise ContractError("acceptance id and description must be non-empty strings") + if not isinstance(item["evidence_kind"], str) or item["evidence_kind"] not in {"review", "check", "artifact", "host"}: raise ContractError("invalid evidence_kind") if not isinstance(value["checks"], list): raise ContractError("checks must be an array") for check in value["checks"]: _exact(check, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, "check") + if not isinstance(check["id"], str) or not check["id"]: raise ContractError("check id must be non-empty") if not isinstance(check["argv"], list) or not check["argv"] or not all(isinstance(v, str) and v for v in check["argv"]): raise ContractError("check argv must be a non-empty string array") _relative(check["cwd"], "check cwd") if not isinstance(check["timeout_seconds"], int) or isinstance(check["timeout_seconds"], bool) or check["timeout_seconds"] <= 0: raise ContractError("check timeout must be positive") if type(check["required_to_pass"]) is not bool: raise ContractError("required_to_pass must be boolean") scope = value["scope"]; _exact(scope, {"read_paths", "write_paths"}, {"read_paths", "write_paths"}, "scope") - if not isinstance(scope["read_paths"], list) or not isinstance(scope["write_paths"], list) or not all(isinstance(p, str) for p in scope["read_paths"] + scope["write_paths"]): raise ContractError("scope paths must be string arrays") + if not isinstance(scope["read_paths"], list) or not isinstance(scope["write_paths"], list) or not all(isinstance(p, str) and p for p in scope["read_paths"] + scope["write_paths"]): raise ContractError("scope paths must be non-empty string arrays") + if len(set(scope["read_paths"])) != len(scope["read_paths"]) or len(set(scope["write_paths"])) != len(scope["write_paths"]): raise ContractError("scope paths must be unique") for p in scope["read_paths"] + scope["write_paths"]: _relative(p, "scope path") if value["workflow"] == "branch-review" and scope["write_paths"]: raise ContractError("branch review cannot write") lead = value["lead"]; _exact(lead, {"mode"}, {"mode"}, "lead") - if lead["mode"] not in {"host", "headless"}: raise ContractError("invalid lead mode") + if not isinstance(lead["mode"], str) or lead["mode"] not in {"host", "headless"}: raise ContractError("invalid lead mode") routing = value["routing"]; _exact(routing, {"profiles_file", "policy_file", "overrides"}, {"profiles_file", "policy_file"}, "routing") for key in ("profiles_file", "policy_file"): if not isinstance(routing[key], str) or not routing[key]: raise ContractError(f"routing {key} must be a path") @@ -60,13 +66,15 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) if role not in {"implementer", "reviewer", "lead", "researcher"}: raise ContractError("invalid override role") _exact(override, {"profile_id", "fallback"}, {"profile_id"}, "routing override") if not isinstance(override["profile_id"], str) or not override["profile_id"]: raise ContractError("override profile_id must be non-empty") - if override.get("fallback", "none") not in {"none", "policy"}: raise ContractError("override fallback must be none or policy") + if not isinstance(override.get("fallback", "none"), str) or override.get("fallback", "none") not in {"none", "policy"}: raise ContractError("override fallback must be none or policy") origin = value["origin"]; _exact(origin, {"surface", "session_ref"}, {"surface"}, "origin") if not isinstance(origin["surface"], str) or not origin["surface"]: raise ContractError("origin surface must be non-empty") + if "session_ref" in origin and (not isinstance(origin["session_ref"], str) or not origin["session_ref"]): raise ContractError("origin session_ref must be non-empty") if "review" in value: review = value["review"]; _exact(review, {"mode", "focus"}, {"mode"}, "review") - if review["mode"] not in {"standard", "adversarial"}: raise ContractError("invalid review mode") + if not isinstance(review["mode"], str) or review["mode"] not in {"standard", "adversarial"}: raise ContractError("invalid review mode") if "focus" in review and review["mode"] != "adversarial": raise ContractError("review focus requires adversarial mode") + if "focus" in review and (not isinstance(review["focus"], str) or not review["focus"]): raise ContractError("review focus must be non-empty") budget = value["budget"]; required = {"wall_seconds", "max_worker_invocations", "max_revisions", "max_fallbacks_per_step"}; _exact(budget, required, required, "budget") for key, number in budget.items(): if not isinstance(number, int) or isinstance(number, bool) or number < 0: raise ContractError(f"budget {key} must be a finite non-negative integer") @@ -80,12 +88,12 @@ def validate_profile(value: dict[str, Any]) -> None: if not isinstance(value[key], str) or not value[key]: raise ContractError(f"profile {key} must be non-empty") effort = value["effort"]; _exact(effort, {"value", "transport"}, {"value", "transport"}, "profile effort") if effort["value"] is not None and not isinstance(effort["value"], str): raise ContractError("effort value must be string or null") - if effort["transport"] not in {"native", "model_variant", "provider_default"}: raise ContractError("invalid effort transport") + if not isinstance(effort["transport"], str) or effort["transport"] not in {"native", "model_variant", "provider_default"}: raise ContractError("invalid effort transport") for key in ("required_tools", "evidence_refs"): - if not isinstance(value[key], list) or not all(isinstance(v, str) for v in value[key]) or len(set(value[key])) != len(value[key]): raise ContractError(f"profile {key} must contain unique strings") - if value["permission_policy"] not in {"read_only", "workspace_write"}: raise ContractError("invalid permission policy") - if value["billing_mode"] not in {"subscription", "paid_api"}: raise ContractError("invalid billing mode") - if value["quality_status"] not in {"unvalidated", "trial", "proven", "suspended"}: raise ContractError("invalid quality status") + if not isinstance(value[key], list) or not all(isinstance(v, str) and v for v in value[key]) or len(set(value[key])) != len(value[key]): raise ContractError(f"profile {key} must contain unique non-empty strings") + if not isinstance(value["permission_policy"], str) or value["permission_policy"] not in {"read_only", "workspace_write"}: raise ContractError("invalid permission policy") + if not isinstance(value["billing_mode"], str) or value["billing_mode"] not in {"subscription", "paid_api"}: raise ContractError("invalid billing mode") + if not isinstance(value["quality_status"], str) or value["quality_status"] not in {"unvalidated", "trial", "proven", "suspended"}: raise ContractError("invalid quality status") def validate_policy(value: dict[str, Any]) -> None: @@ -93,12 +101,36 @@ def validate_policy(value: dict[str, Any]) -> None: required = fields - {"prefer_different_harness_for_review"} _exact(value, fields, required, "policy") if type(value["schema_version"]) is not int or value["schema_version"] != 1 or type(value["version"]) is not int or value["version"] < 1: raise ContractError("invalid policy version") + if not isinstance(value["id"], str) or not value["id"]: raise ContractError("policy id must be non-empty") if type(value["require_different_model_for_review"]) is not bool or ("prefer_different_harness_for_review" in value and type(value["prefer_different_harness_for_review"]) is not bool): raise ContractError("policy review flags must be boolean") if not isinstance(value["roles"], dict) or set(value["roles"]) - {"implementer", "reviewer", "lead", "researcher"}: raise ContractError("invalid policy roles") for candidates in value["roles"].values(): if not isinstance(candidates, list) or not candidates: raise ContractError("role candidates must be non-empty arrays") for ref in candidates: _exact(ref, {"kind", "id"}, {"kind", "id"}, "candidate reference") - if ref["kind"] not in {"profile", "alias"} or not isinstance(ref["id"], str) or not ref["id"]: raise ContractError("invalid candidate reference") + if not isinstance(ref["kind"], str) or ref["kind"] not in {"profile", "alias"} or not isinstance(ref["id"], str) or not ref["id"]: raise ContractError("invalid candidate reference") for key in ("task_classes", "account_pools", "experiment_budget"): if not isinstance(value[key], dict): raise ContractError(f"policy {key} must be an object") + _validate_json_tree(value[key], f"policy {key}") + if not all(isinstance(k, str) and k and isinstance(v, str) and v in {"unvalidated", "trial", "proven", "suspended"} for k, v in value["task_classes"].items()): + raise ContractError("task_classes must map names to quality status") + if not all(isinstance(k, str) and k and isinstance(v, dict) for k, v in value["account_pools"].items()): + raise ContractError("account_pools must map names to objects") + if not all(isinstance(k, str) and k and type(v) is int and v >= 0 for k, v in value["experiment_budget"].items()): + raise ContractError("experiment_budget must contain non-negative integers") + + +def _validate_json_tree(value: Any, label: str) -> None: + """Reject non-JSON and numerically ambiguous values in extension maps.""" + if value is None or isinstance(value, str) or type(value) is bool: + return + if type(value) is int: + return + if isinstance(value, list): + for item in value: _validate_json_tree(item, label) + return + if isinstance(value, dict): + if not all(isinstance(k, str) and k for k in value): raise ContractError(f"{label} keys must be non-empty strings") + for item in value.values(): _validate_json_tree(item, label) + return + raise ContractError(f"{label} contains a non-JSON or non-finite value") diff --git a/plugin/lib/adapter.sh b/plugin/lib/adapter.sh index 45dba9b..2a212d2 100644 --- a/plugin/lib/adapter.sh +++ b/plugin/lib/adapter.sh @@ -32,6 +32,8 @@ set -euo pipefail _ADAPTER_LIB_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # shellcheck source=/dev/null source "${_ADAPTER_LIB_DIR}/model-catalog.sh" +# shellcheck source=/dev/null +source "${_ADAPTER_LIB_DIR}/../core/adapters/classification-policy.conf" # Terminate a bounded subprocess tree without requiring GNU timeout, setsid, # or job-control process groups (all absent on a stock macOS Bash 3.2 host). @@ -161,30 +163,34 @@ _adapter_invoke() { "$cli" "${ADAPTER_ARGS[@]}" >"$stdout_file" 2>"$stderr_file" & fi local cli_pid=$! - local timed_out_file="${stdout_file}.timed-out" - local process_snapshot="${stdout_file}.processes" - local watchdog_snapshot="${stdout_file}.watchdog-processes" - # Redirect the watchdog itself: inherited capture descriptors were the - # reason a successful immediate command waited for the full timeout. - ( sleep "$timeout_secs"; _adapter_snapshot_tree "$cli_pid" > "$process_snapshot"; : > "$timed_out_file"; _adapter_signal_snapshot "$process_snapshot" TERM; sleep 0.1; _adapter_signal_snapshot "$process_snapshot" KILL ) >/dev/null 2>&1 & - local watchdog_pid=$! + local process_snapshot="${stdout_file}.processes" timed_out="false" + local polls_remaining=$(( timeout_secs * 20 )) state="" + # Poll the directly-owned child. This avoids a background sleep/watchdog + # retaining capture descriptors after fast completion. + while :; do + state=$(ps -p "$cli_pid" -o stat= 2>/dev/null | tr -d ' ' || true) + [[ -z "$state" || "$state" == Z* ]] && break + if [[ "$polls_remaining" -le 0 ]]; then + timed_out="true" + _adapter_snapshot_tree "$cli_pid" > "$process_snapshot" + _adapter_signal_snapshot "$process_snapshot" TERM + sleep 0.1 + _adapter_signal_snapshot "$process_snapshot" KILL + break + fi + sleep 0.05 + polls_remaining=$(( polls_remaining - 1 )) + done if wait "$cli_pid"; then exit_code=0 else exit_code=$? fi - # Snapshot the watchdog subtree while its sleep child is still attached, - # then stop the exact identities. This never scans after orphaning. - _adapter_snapshot_tree "$watchdog_pid" > "$watchdog_snapshot" - _adapter_signal_snapshot "$watchdog_snapshot" KILL - wait "$watchdog_pid" 2>/dev/null || true - if [[ -f "$timed_out_file" ]]; then - # The watchdog performs escalation before wait can return. Repeat the - # exact captured set defensively; never discover unrelated PIDs here. + if [[ "$timed_out" == "true" ]]; then _adapter_signal_snapshot "$process_snapshot" KILL exit_code=124 fi - rm -f "$timed_out_file" "$process_snapshot" "$watchdog_snapshot" + rm -f "$process_snapshot" fi local stdout stderr_content @@ -207,9 +213,9 @@ _adapter_invoke() { return 0 elif [[ $exit_code -eq 124 ]]; then _adapter_fail "TIMEOUT: ${agent} did not respond within ${timeout_secs}s. ${ADAPTER_FALLBACK}" - elif echo "$stderr_content" | grep -qiE 'auth|401|403|ineligible|unauthorized'; then + elif echo "$stderr_content" | grep -qiE "$DEVSQUAD_AUTH_ERROR_PATTERN"; then _adapter_fail "AUTH_ERROR: ${agent} CLI authentication failed. ${ADAPTER_AUTH_HINT}" - elif echo "$stderr_content" | grep -qiE '429|rate.?limit|quota|resource.?exhausted|too many requests'; then + elif echo "$stderr_content" | grep -qiE "$DEVSQUAD_RATE_LIMIT_PATTERN"; then record_rate_limit "$state_dir" "$agent" _adapter_fail "RATE_LIMITED: ${agent} hit a rate limit. 2-minute cooldown started. ${ADAPTER_FALLBACK}" else diff --git a/test/core/fakes/codex_app_server.py b/test/core/fakes/codex_app_server.py index 3a61815..9722d57 100644 --- a/test/core/fakes/codex_app_server.py +++ b/test/core/fakes/codex_app_server.py @@ -1,16 +1,23 @@ #!/usr/bin/env python3 import json, sys, time +initialized = False for line in sys.stdin: request = json.loads(line) method = request.get("method") if method == "initialize": print(json.dumps({"id":request["id"],"result":{"serverInfo":{"name":"fake","version":"1"}}}), flush=True) + elif method == "initialized": + initialized = True elif method == "model/list": + if not initialized: + print(json.dumps({"id":request["id"],"error":{"code":-32002,"message":"not initialized"}}), flush=True); continue cursor = request.get("params",{}).get("cursor") result = {"data":[{"id":"gpt-fake","supportedReasoningEfforts":[{"reasoningEffort":"low","description":"fixture"}]}],"nextCursor":"two"} if cursor is None else {"data":[],"nextCursor":None} print(json.dumps({"method":"account/updated","params":{"reason":"fixture"}}), flush=True) print(json.dumps({"id":request["id"],"result":result}), flush=True) elif method == "turn/start": + if not initialized: + print(json.dumps({"id":request["id"],"error":{"code":-32002,"message":"not initialized"}}), flush=True); continue print(json.dumps({"id":request["id"],"result":{"turn":{"id":"turn-1"}}}), flush=True) time.sleep(0.02) print(json.dumps({"method":"turn/started","params":{"threadId":request["params"]["threadId"],"turn":{"id":"turn-1","status":"inProgress"}}}), flush=True) diff --git a/test/core/test_m1.py b/test/core/test_m1.py index 0a502b0..eb5fbcf 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -15,7 +15,7 @@ from devsquad.adapters import AdapterManifest, classify_cli, prepare_cli, prepare_native_codex, prepare_native_codex_from_catalog from devsquad.catalog import update_last_good -from devsquad.codex_protocol import JsonLinePeer, NativeTurnState, collect_model_pages, discover_models, initialize_request, model_list_request, parse_model_page, receive_response, review_start_request, thread_start_request, turn_interrupt_request, turn_start_request +from devsquad.codex_protocol import JsonLinePeer, NativeTurnState, collect_model_pages, discover_models, initialize_request, initialized_notification, model_list_request, parse_model_page, receive_response, review_start_request, thread_start_request, turn_interrupt_request, turn_start_request from devsquad.contracts import ContractError, validate_launch_payload from devsquad.validation import validate_policy, validate_profile, validate_task @@ -56,6 +56,8 @@ def test_classifier_does_not_accept_empty_denied_or_malformed_exit_zero(self): self.assertEqual(classify_cli(spec, returncode=0, stdout='{"type":"result","is_error":true,"error":"tool denied"}', stderr="").execution_status, "denied") self.assertEqual(classify_cli(spec, returncode=0, stdout="not json", stderr="").execution_status, "malformed") self.assertEqual(classify_cli(spec, returncode=0, stdout="42", stderr="").execution_status, "malformed") + nested_bad = '{"type":"item.completed","item":"bad"}\n{"type":"turn.completed"}' + self.assertEqual(classify_cli(spec, returncode=0, stdout=nested_bad, stderr="").execution_status, "malformed") def test_auth_precedes_rate_and_acceptance_is_separate(self): temp, binary = self.fake_path("grok") @@ -95,6 +97,7 @@ def test_native_launch_is_preparation_only_and_version_scoped(self): spec = prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.135.0") self.assertEqual(spec.transport, "native_protocol") self.assertEqual(spec.argv[-2:], ("--listen", "stdio://")) + self.assertIn('model_reasoning_effort="low"', spec.argv) self.assertEqual(spec.environment["DEVSQUAD_WORKER"], "1") with self.assertRaises(ContractError): prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli future") @@ -165,7 +168,7 @@ def cleanup(): process.stdin.close(); process.stdout.close() self.addCleanup(cleanup) peer = JsonLinePeer(process.stdout, process.stdin) - peer.send(initialize_request(1)); self.assertIn("result", receive_response(peer, 1, timeout_seconds=2)) + peer.send(initialize_request(1)); self.assertIn("result", receive_response(peer, 1, timeout_seconds=2)); peer.send(initialized_notification()) self.assertEqual([m["id"] for m in discover_models(peer, first_request_id=2, timeout_seconds=2)], ["gpt-fake"]) peer.send(turn_start_request(4, thread_id="thread-1", prompt="p", model="gpt-fake", effort="low", cwd="/tmp", permission="read_only")) response = receive_response(peer, 4, timeout_seconds=2); self.assertEqual(response["result"]["turn"]["id"], "turn-1") diff --git a/test/core/test_m1_gate_review.py b/test/core/test_m1_gate_review.py new file mode 100644 index 0000000..db155ff --- /dev/null +++ b/test/core/test_m1_gate_review.py @@ -0,0 +1,140 @@ +"""Independent M1 review regressions for externally supplied data and framing. + +These cases came from reviewing the first implementation, rather than from +its internal structure. They run without provider access or a core install. +""" +from __future__ import annotations + +import copy +import json +import os +from pathlib import Path +import sys +import threading +import time +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) + +from devsquad.adapters import classify_cli +from devsquad.codex_protocol import JsonLinePeer +from devsquad.contracts import ( + ContractError, ExecutionIdentity, LaunchSpec, validate_launch_payload, +) +from devsquad.validation import validate_task + + +def launch() -> LaunchSpec: + return LaunchSpec( + 1, "codex", "cli_exec", ("fake-codex",), "/tmp", None, 2, + ExecutionIdentity("codex", None, None, None, None, None), + ) + + +class UntrustedInputReview(unittest.TestCase): + def test_launch_rejects_invalid_types_without_coercion(self): + cases = [ + ("schema boolean", {"schema_version": True}), + ("timeout boolean", {"timeout_seconds": True}), + ("timeout NaN", {"timeout_seconds": float("nan")}), + ("timeout infinity", {"timeout_seconds": float("inf")}), + ("timeout fraction", {"timeout_seconds": 1.5}), + ("string argv", {"argv": "fake-codex"}), + ("object argv", {"argv": {"fake-codex": True}}), + ("empty argv", {"argv": []}), + ("relative cwd", {"cwd": "relative"}), + ("arbitrary environment", {"environment": {"UNDECLARED_FLAG": "1"}}), + ] + for name, change in cases: + with self.subTest(name=name): + value = launch().to_dict() + value.update(change) + with self.assertRaises(ContractError): + validate_launch_payload(value) + + def test_task_rejects_invalid_nested_values(self): + original = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + cases = [ + ("base ref", ("project", "base_ref"), {}), + ("criterion id", ("acceptance", 0, "id"), {}), + ("criterion description", ("acceptance", 0, "description"), []), + ("check id", ("checks", 0, "id"), 123), + ("check argv", ("checks", 0, "argv"), "echo hello"), + ("check boolean", ("checks", 0, "required_to_pass"), "false"), + ("session reference", ("origin", "session_ref"), {}), + ("scope list", ("scope", "read_paths"), "src"), + ("finite budget", ("budget", "wall_seconds"), float("nan")), + ("fallback", ("routing", "overrides"), { + "reviewer": {"profile_id": "fixture", "fallback": "anything"}, + }), + ] + for name, path, value in cases: + with self.subTest(name=name): + task = copy.deepcopy(original) + target = task + for key in path[:-1]: + target = target[key] + target[path[-1]] = value + with self.assertRaises(ContractError): + validate_task(task) + + def test_malformed_provider_frames_return_a_verdict(self): + for frame in (42, None, ["bad"], {"type": "item.completed", "item": "bad"}, + {"type": "item.completed", "item": 7}): + with self.subTest(frame=frame): + result = classify_cli(launch(), returncode=0, stdout=json.dumps(frame), stderr="") + self.assertEqual(result.error_code, "CLI_ERROR") + self.assertNotEqual(result.execution_status, "succeeded") + + def test_partial_native_message_is_not_completion(self): + frame = {"type": "item.completed", "item": {"type": "agent_message", "text": "partial"}} + result = classify_cli(launch(), returncode=0, stdout=json.dumps(frame), stderr="") + self.assertEqual(result.error_code, "CLI_ERROR") + self.assertNotEqual(result.execution_status, "succeeded") + + +class NativeFramingReview(unittest.TestCase): + def test_two_frames_in_one_write_are_both_available(self): + read_fd, write_fd = os.pipe() + with os.fdopen(read_fd, "r") as reader, os.fdopen(write_fd, "w") as writer: + peer = JsonLinePeer(reader, writer) + writer.write('{"n":1}\n{"n":2}\n') + writer.flush() + self.assertEqual(peer.receive(0.2), {"n": 1}) + self.assertEqual(peer.receive(0.2), {"n": 2}) + + def test_incomplete_frame_respects_receive_deadline(self): + read_fd, write_fd = os.pipe() + with os.fdopen(read_fd, "r") as reader, os.fdopen(write_fd, "w") as writer: + peer = JsonLinePeer(reader, writer) + writer.write("{") + writer.flush() + outcomes = [] + + def receive(): + try: + outcomes.append(peer.receive(0.05)) + except Exception as exc: + outcomes.append(exc) + + worker = threading.Thread(target=receive, daemon=True) + started = time.monotonic() + worker.start() + worker.join(0.4) + exceeded_deadline = worker.is_alive() + # Release a buggy blocking readline before asserting, so a failed + # regression does not leave a test thread or pipe behind. + if exceeded_deadline: + writer.write('"late":true}\n') + writer.flush() + worker.join(1) + self.assertFalse(exceeded_deadline, "receive ignored its deadline") + self.assertLess(time.monotonic() - started, 0.4) + self.assertIsInstance(outcomes[0], TimeoutError) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_validation.py b/test/core/test_validation.py new file mode 100644 index 0000000..0b60328 --- /dev/null +++ b/test/core/test_validation.py @@ -0,0 +1,69 @@ +from __future__ import annotations + +import copy +import json +import math +import sys +import unittest +from pathlib import Path + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.contracts import ContractError, ExecutionIdentity, LaunchSpec, NormalizedResult, validate_launch_payload +from devsquad.validation import validate_policy, validate_profile, validate_task + + +class AdversarialValidationTest(unittest.TestCase): + def setUp(self) -> None: + root = Path(__file__).resolve().parents[2] + self.task = json.loads((root / "docs/plans/engineering-team/examples/issue-delivery.json").read_text()) + self.identity = ExecutionIdentity("codex", None, "openai", "gpt", "gpt-test", "low", ("read",), "read_only", "pool", "verified") + + def assert_contract_error(self, fn, *args): + with self.assertRaises(ContractError): + fn(*args) + + def test_launch_rejects_bool_nan_string_argv_and_environment_abuse(self): + base = LaunchSpec(1, "codex", "cli_exec", ("codex",), "/tmp", None, 3, self.identity).to_dict() + for key, bad in (("timeout_seconds", True), ("timeout_seconds", math.nan), ("argv", "codex exec")): + value = copy.deepcopy(base); value[key] = bad + self.assert_contract_error(validate_launch_payload, value) + for env in ({"PATH": "/tmp"}, {"DEVSQUAD_WORKER": True}): + value = copy.deepcopy(base); value["environment"] = env + self.assert_contract_error(validate_launch_payload, value) + + def test_identity_and_result_validate_all_nested_fields(self): + for change in ({"harness_version": 1}, {"model": ""}, {"tools": ["read", "read"]}, {"account_pool": {}}): + raw = {**self.identity.__dict__, **change} + raw["tools"] = tuple(raw["tools"]) + with self.assertRaises(ContractError): + ExecutionIdentity(**raw) + self.assert_contract_error(NormalizedResult, 1, "succeeded", None, None, "unknown", "not_evaluated", self.identity, None, {"turn": 1}) + + def test_task_rejects_adversarial_nested_types(self): + mutations = [ + lambda t: t["acceptance"][0].__setitem__("id", {"nested": "id"}), + lambda t: t["checks"][0].__setitem__("argv", "python -m test"), + lambda t: t["checks"][0].__setitem__("timeout_seconds", True), + lambda t: t["budget"].__setitem__("wall_seconds", True), + lambda t: t["scope"]["read_paths"].append("../escape"), + lambda t: t["origin"].__setitem__("session_ref", []), + ] + for mutate in mutations: + value = copy.deepcopy(self.task); mutate(value) + self.assert_contract_error(validate_task, value) + + def test_profile_and_policy_reject_nested_type_confusion(self): + profile = {"id":"p","harness":"codex","model_family":"gpt","model_id":"m","effort":{"value":"low","transport":"native"},"required_tools":["read"],"permission_policy":"read_only","account_pool_id":"pool","billing_mode":"subscription","quality_status":"proven","evidence_refs":[]} + bad = copy.deepcopy(profile); bad["required_tools"] = [""] + self.assert_contract_error(validate_profile, bad) + policy = {"schema_version":1,"id":"p","version":1,"roles":{"reviewer":[{"kind":"profile","id":"p"}]},"task_classes":{},"require_different_model_for_review":True,"account_pools":{},"experiment_budget":{}} + validate_policy(policy) + for field, value in (("version", True), ("id", {}), ("experiment_budget", {"limit": math.nan})): + bad = copy.deepcopy(policy); bad[field] = value + self.assert_contract_error(validate_policy, bad) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/test_m1_legacy.sh b/test/test_m1_legacy.sh index 9d1db10..f715b89 100755 --- a/test/test_m1_legacy.sh +++ b/test/test_m1_legacy.sh @@ -41,10 +41,17 @@ chmod +x "$T/bin/sleep" # Force the portable path even on systems with timeout/gtimeout installed. WATCHDOG_SLEEP_PID_FILE="$T/watchdog-sleep.pid"; export WATCHDOG_SLEEP_PID_FILE FAST_ELAPSED_FILE="$T/fast-elapsed"; export FAST_ELAPSED_FILE -HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=fast \ - bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; s=$(perl -MTime::HiRes=time -e '\''printf "%.6f", time'\''); invoke_codex hello 10 2; f=$(perl -MTime::HiRes=time -e '\''printf "%.6f", time'\''); awk -v s="$s" -v f="$f" '\''BEGIN { printf "%.3f", f-s }'\'' > "$FAST_ELAPSED_FILE"' _ "$ROOT" > "$T/out" 2> "$T/err" +NOW_BIN="$T/bin/now"; export NOW_BIN +cat > "$NOW_BIN" <<'EOF' +#!/usr/bin/perl +use Time::HiRes qw(time); +printf "%.6f", time; +EOF +chmod +x "$NOW_BIN" +HOME="$T/home" PATH="/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="/bin/echo" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=fast \ + bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; s=$($NOW_BIN); invoke_codex hello 10 2; f=$($NOW_BIN); awk -v s="$s" -v f="$f" '\''BEGIN { printf "%.3f", f-s }'\'' > "$FAST_ELAPSED_FILE"' _ "$ROOT" > "$T/out" 2> "$T/err" fast_elapsed=$(cat "$FAST_ELAPSED_FILE") -[[ "$(cat "$T/out")" == "ok" ]] && ok || bad "portable watchdog output" +grep -q 'exec hello' "$T/out" && ok || bad "portable watchdog output" awk -v e="$fast_elapsed" 'BEGIN { exit !(e < 1.0) }' && ok || bad "portable fast call took ${fast_elapsed}s" if [[ -s "$WATCHDOG_SLEEP_PID_FILE" ]] && kill -0 "$(cat "$WATCHDOG_SLEEP_PID_FILE")" 2>/dev/null; then bad "portable success left watchdog sleep alive" From cc98ce77e2f7027c63eca1b3b2a0d5c120727e7b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:13:08 +0530 Subject: [PATCH 012/197] docs: record M1 candidate gate evidence --- docs/plans/engineering-team/M1-STATUS.md | 49 ++++++++++++------- docs/plans/engineering-team/backlog.json | 10 ++-- .../M1-invocation-core-2026-09-06.json | 26 ++++++---- 3 files changed, 52 insertions(+), 33 deletions(-) diff --git a/docs/plans/engineering-team/M1-STATUS.md b/docs/plans/engineering-team/M1-STATUS.md index a486da7..41c5318 100644 --- a/docs/plans/engineering-team/M1-STATUS.md +++ b/docs/plans/engineering-team/M1-STATUS.md @@ -1,22 +1,35 @@ # M1 implementation status -M1 is implemented at checkpoint `572e452` and remains **in progress** pending -independent review of the complete gate. M2 process ownership has not started. +M1 is implemented through candidate `a67ab58` and remains **in progress** +pending independent review and a successful integrated native live probe. M2 +process ownership has not started. -| Requirement | Evidence | Status | -|---|---|---| -| Python 3.11+ package, launcher, strict v1 inputs | `plugin/core`, 13 standard-library unit tests, source and installed-wheel launch | verified | -| Requested and observed identity remain separate | `ExecutionIdentity`, `LaunchSpec`, `NormalizedResult` | verified offline | -| Execution, artifact and acceptance states remain separate | normalized-result schema and classifier tests | verified offline | -| Explicit model/effort/permission preparation | manifest builders reject unknown effort pairs; argv preserves spaces | verified offline | -| Codex native metadata and lifecycle dialect | installed 0.135.0 schema generation, live paginated `model/list`, fake event-state tests | verified | -| Catalog last-good retention and unqualified discovery | catalog unit and legacy shell tests | verified offline | -| Legacy wrapper API and four error prefixes | existing discovery suite, 38 wrapper assertions | verified offline | -| Portable watchdog prompt return and descendant cleanup | forced portable-path Bash 3.2 tests | verified offline | -| Tracked bounded Antigravity context | Git inventory tests cover spaces, TSX/JSX, ignored/binary/symlink/escape/size omissions | verified offline | -| Bounded real starting-profile smoke | Codex `gpt-5.5`, low effort, read-only, ephemeral JSONL invocation | verified live | -| Independent gate review | Root review is active; findings were folded into the checkpoint | pending final review | +| # | Requirement | Evidence | Status | +|---|---|---|---| +| 1 | Native framing, typed requests, handshake, lifecycle and correlation | Bounded buffered JSON-line peer; `initialize` then `initialized`; installed 0.135.0 request shapes; conforming fake server with interleaved notifications | verified offline | +| 2 | Truthful provider completion and malformed-frame handling | Codex requires correlated terminal `turn/completed`; provider-specific document/JSONL parsing; partial, startup-only and malformed fixtures | verified offline | +| 3 | Strict Task, Profile, Policy, LaunchSpec, identity and result inputs | Runtime validators, matching strict schemas, adversarial nested-type tests and independent reviewer regressions | verified offline | +| 4 | Model/version-scoped capability preparation | Complete paginated discovery feeds native preparation; explicit unknown model, effort and drift cases fail before launch | verified offline | +| 5 | Shared Bash/Python classification taxonomy | Packaged `classification-policy.conf` is consumed by both paths; auth-topic, ordering and empty-success regressions | verified offline | +| 6 | Portable watchdog ownership and cleanup | Forced portable Bash 3.2 tests cover fast return, descendant cleanup, and a ready TERM-ignoring root with an absolute fake executable | verified offline | +| 7 | Complete native pagination and last-good behavior | Empty pages, repeated cursors, malformed/disconnected pages and incomplete-refresh retention are exercised | verified offline | +| 8 | Requested and observed identity remain separate | `ExecutionIdentity`, `LaunchSpec` and `NormalizedResult` validation and round trips | verified offline | +| 9 | Execution, artifact and acceptance states remain separate | Normalized result contract and classifier tests | verified offline | +| 10 | Tracked, bounded Antigravity context | Literal Git inventory; ignored/binary/secret/oversize omissions; file and ancestor-symlink containment tests | verified offline | +| 11 | Package and CLI envelope | Temporary wheel installation resolves schemas, adapters and shared taxonomy; input/readiness exit behavior is tested | verified offline | +| 12 | Bounded real starting-profile smoke | Codex CLI `gpt-5.5`, low effort, read-only, ephemeral JSONL invocation completed on 2026-09-06 | verified live (CLI) | +| 13 | Integrated native preparation/protocol/classification probe | Correct handshake initializes, but the earlier inline app-server attempt stalled at `model/list` or `thread/start`; root is diagnosing with a saved probe | pending live | +| 14 | Independent gate review | Reviewer regressions are tracked and pass; final root disposition remains outstanding | pending final review | -The Python bridge returns argv and parser policy only. It does not spawn a -worker or implement a watchdog. M2 remains the sole owner of process sessions, -timeouts, cancellation, draining and reaping for new runs. +The Python bridge returns launch/protocol preparation and normalized parser +policy only. It does not spawn a worker or implement a second watchdog. M2 +remains the sole owner of process sessions, timeouts, cancellation, draining +and reaping for new runs. + +Declared limitations at this checkpoint: + +- Grok workspace-write preparation is unsupported; its verified profile is + read-only. +- Antigravity and Grok inference were not used for the live M1 smoke. +- Policy learning, routing semantics and session supervision remain deferred + to their specified later milestones. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 62e71c0..eeb4c0b 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -22,15 +22,15 @@ "evidence": [ { "kind": "implementation_checkpoint", - "revision": "572e452", - "command_or_action": "offline core/legacy suites, temporary wheel install, live Codex app-server metadata and read-only invocation probe", - "outcome": "M1 implementation and stated probes pass; independent gate review remains open", + "revision": "a67ab58", + "command_or_action": "32 core tests, 202 shell assertions, temporary wheel install, prior live Codex metadata and read-only CLI smoke", + "outcome": "All enumerated offline M1 gates pass; saved integrated native live probe and independent final review remain open", "artifact": "evidence/M1-invocation-core-2026-09-06.json", - "recorded_at": "2026-09-06T13:02:37Z", + "recorded_at": "2026-09-07T08:20:00+05:30", "availability": "portable_redacted" } ], - "blocker": null + "blocker": "Integrated native app-server probe and independent final M1 review are pending." }, { "id": "M2", diff --git a/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json b/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json index fc16268..83b87ea 100644 --- a/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json +++ b/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json @@ -2,42 +2,48 @@ "schema_version": 1, "milestone": "M1", "status": "in_progress", - "implementation_revision": "572e452", - "implementation_tree": "26abe92ee4c89dbd481ada20ebd02fb9d6898f51", - "recorded_at": "2026-09-06T13:02:37Z", + "implementation_revision": "a67ab58", + "recorded_at": "2026-09-07T08:20:00+05:30", "evidence": [ { "kind": "offline_test", "command_or_action": "/bin/bash test/run.sh", - "outcome": "10 test files passed; 192 shell assertions passed including Bash 3.2 portable timeout and descendant cleanup", + "outcome": "10 test files passed; 202 shell assertions passed, including forced portable timeout, fast cleanup, descendant cleanup and TERM-ignoring-root escalation", "availability": "tracked tests" }, { "kind": "offline_test", "command_or_action": "PYTHONDONTWRITEBYTECODE=1 python3 -m unittest discover -s test/core -v", - "outcome": "13 tests passed", + "outcome": "32 tests passed, including six independent reviewer regressions", "availability": "tracked tests" }, { "kind": "package_test", - "command_or_action": "bundled Python 3.12: pip wheel --no-build-isolation --no-deps plugin/core; install into temporary venv; squad --version; squad doctor --json", - "outcome": "wheel built and installed; packaged adapter resources resolved outside checkout; squad 0.1.0 reported installed harnesses", + "command_or_action": "bundled Python 3.12: pip wheel --no-deps plugin/core; install into temporary venv; import packaged classifier and resolve shared policy resource", + "outcome": "wheel built and installed; shared taxonomy loaded outside the checkout", "availability": "reproducible locally" }, { "kind": "live_metadata", - "command_or_action": "Codex 0.135.0 app-server initialize then model/list with limit 100", + "command_or_action": "Codex 0.135.0 app-server initialize then model/list with limit 100 on 2026-09-06", "outcome": "initialized; complete single page with 5 account-visible models and model-scoped reasoning efforts; default gpt-5.5/medium", "availability": "redacted portable summary" }, { "kind": "live_smoke", - "command_or_action": "codex exec --json --sandbox read-only --ephemeral --model gpt-5.5 with model_reasoning_effort=low and a no-tools response probe", + "command_or_action": "codex exec --json --sandbox read-only --ephemeral --model gpt-5.5 with model_reasoning_effort=low and a no-tools response probe on 2026-09-06", "outcome": "exit 0; thread.started, turn.started, item.completed and turn.completed observed; final message DEVSQUAD_M1_PROBE_OK", "availability": "redacted portable summary" + }, + { + "kind": "live_integrated_probe", + "command_or_action": "production preparation and JSON-line peer: initialize response, initialized notification, then model/list or prepared thread/start; bounded per-process low-effort override", + "outcome": "Earlier inline attempt initialized, then did not receive model/list or thread/start within its bound. Root is rerunning a saved diagnostic probe, including stderr draining; no integrated success is claimed yet", + "availability": "pending saved diagnostic evidence" } ], "residual_blockers": [ - "Independent M1 gate review is not yet complete." + "The integrated native app-server path has not yet completed a saved real turn probe.", + "Independent final M1 gate review is pending." ] } From ad46b2fcf1ad61245a4801da60d7581f35142809 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:15:50 +0530 Subject: [PATCH 013/197] docs: save quota recovery checkpoint and exact next steps --- docs/plans/engineering-team/RESUME.md | 59 +++++++++++++++++++++++++++ 1 file changed, 59 insertions(+) create mode 100644 docs/plans/engineering-team/RESUME.md diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md new file mode 100644 index 0000000..598cbe9 --- /dev/null +++ b/docs/plans/engineering-team/RESUME.md @@ -0,0 +1,59 @@ +# Resume DevSquad after an interruption + +This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. + +## Current position — September 7, 2026 + +- Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. +- Build branch: `codex/engineering-team`. `main` is the published runtime baseline. +- Last verified implementation checkpoint: `a67ab58`. +- Last evidence/status checkpoint before this recovery note: `cc98ce7`. +- GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. +- User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. +- Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 has not started. + +## Completed and preserved + +Branch cleanup is complete. Local/GitHub working branch names were consolidated into `main` and `codex/engineering-team`. The old assessment and holdout commits remain in their descendant histories. The unrelated February backup is preserved by its existing local tag and a verified complete Git bundle. See the [branch record](../../audits/2026-09-06-branch-consolidation.md). + +The M1 implementation includes Python packaging/contracts, native Codex protocol preparation and framing, catalog-to-profile preparation, shared error-classification policy, strict input validation, and legacy timeout/context/catalog fixes. Earlier review defects have corresponding regression tests, including [the independent review cases](../../../test/core/test_m1_gate_review.py). + +Verified at the implementation/evidence checkpoints above: + +| Check | Result | +|---|---| +| Python core discovery | 32 tests passed | +| Bash 3.2 regression suite | 10 test files, 202 assertions passed | +| Wheel installation | Temporary venv resolves packaged schemas, adapters and shared taxonomy | +| Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | +| Integrated native adapter proof | **Still pending** | + +The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed evidence is [M1-invocation-core-2026-09-06.json](evidence/M1-invocation-core-2026-09-06.json). [backlog.json](backlog.json) retains M1 as `in_progress`. Grok workspace-write and unprobed Antigravity/Grok settings are not advertised as verified. + +## Exact next work + +1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. +2. Save a reproducible, explicitly opt-in native probe script **before** running it. Previous integrated probes were inline scripts and have no retained raw logs, so their reported failures cannot yet be independently diagnosed. +3. Exercise the actual path: initialized app-server → complete model discovery → saved snapshot → prepared LaunchSpec → native thread/turn → correlated terminal result. Use one bounded read-only inference, temporary workspace and existing subscription authentication. Save a redacted result plus private local diagnostic files. +4. Investigate the nonresponse before declaring an external blocker. Prior probes reported successful initialization but no `model/list` or `thread/start` reply after bounded waits, despite sending the required `initialized` notification. They reported an installed Codex 0.135.0 warning about the global `ultra` effort value. Per-process overrides were attempted; global settings were not changed. **A new hypothesis to test is blocked stderr output from an undrained subprocess PIPE.** Redirect stderr to a private file or drain it concurrently; a full stderr pipe can stall a child. This cause is not yet established. +5. If the actual adapter path succeeds, record the exact revision/commands/outcome and complete the remaining independent M1 review. Only then close M1 and proceed to M2. If it fails, retain the exact diagnostics and keep the live gate open; distinguish implementation defects from provider/configuration limitations. + +The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. + +## Checkpoint discipline + +- Commit coherent partial work and its evidence at small intervals; do not wait for an entire milestone. Mark incomplete work accurately. +- Before long probes or a likely usage cutoff, update this recovery note and checkpoint. Keep the working tree clean at a pause; never stash. +- Run the required `bash test/run.sh` before each commit, and the relevant core tests for code changes. Record failing checks when saving a necessary WIP checkpoint rather than calling it complete. +- Keep raw private prompts, credentials and native diagnostic logs outside tracked evidence. Preserve reproducible scripts and redacted receipts in the repository. +- An account limit does not authorize purchasing credits, consuming a reset credit, silently using paid APIs or changing the requested implementation model. Resume when capacity is available or the user supplies new instructions. +- Do not promise execution while the account is blocked. The committed work and this file are the handoff across that interruption. + +```bash +git status --short --branch +git log -6 --oneline +PYTHONDONTWRITEBYTECODE=1 python3 -m unittest discover -s test/core -v +bash test/run.sh +``` + +Continue from the earliest unfinished requirement with available dependencies. Preserve all later implementation and review findings if this note is older than the current branch. From aa3fe1c2005026cb83777fef39d02cfd932dbe12 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:19:53 +0530 Subject: [PATCH 014/197] test: preserve bounded native M1 probe --- docs/plans/engineering-team/RESUME.md | 9 +- test/core/probes/native_codex_smoke.py | 189 +++++++++++++++++++++++++ 2 files changed, 196 insertions(+), 2 deletions(-) create mode 100644 test/core/probes/native_codex_smoke.py diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 598cbe9..32aa57f 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -8,6 +8,8 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne - Build branch: `codex/engineering-team`. `main` is the published runtime baseline. - Last verified implementation checkpoint: `a67ab58`. - Last evidence/status checkpoint before this recovery note: `cc98ce7`. +- Recovery checkpoint: `ad46b2f`. The opt-in saved probe is + `test/core/probes/native_codex_smoke.py`; commit it before its first live run. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 has not started. @@ -33,8 +35,11 @@ The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed e ## Exact next work 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. -2. Save a reproducible, explicitly opt-in native probe script **before** running it. Previous integrated probes were inline scripts and have no retained raw logs, so their reported failures cannot yet be independently diagnosed. -3. Exercise the actual path: initialized app-server → complete model discovery → saved snapshot → prepared LaunchSpec → native thread/turn → correlated terminal result. Use one bounded read-only inference, temporary workspace and existing subscription authentication. Save a redacted result plus private local diagnostic files. +2. Run the committed opt-in probe once with `--run-live`. It writes stdout + protocol frames and stderr to separate private files under + `~/.devsquad/private-probes`, uses bounded deadlines, and records cleanup and + hashes in its private receipt. Previous inline probes have no raw logs. +3. The probe exercises the actual path: initialized app-server → complete model discovery → saved snapshot → prepared LaunchSpec → native thread/turn → correlated terminal result. It uses one bounded read-only inference, a temporary Git workspace and existing subscription authentication. 4. Investigate the nonresponse before declaring an external blocker. Prior probes reported successful initialization but no `model/list` or `thread/start` reply after bounded waits, despite sending the required `initialized` notification. They reported an installed Codex 0.135.0 warning about the global `ultra` effort value. Per-process overrides were attempted; global settings were not changed. **A new hypothesis to test is blocked stderr output from an undrained subprocess PIPE.** Redirect stderr to a private file or drain it concurrently; a full stderr pipe can stall a child. This cause is not yet established. 5. If the actual adapter path succeeds, record the exact revision/commands/outcome and complete the remaining independent M1 review. Only then close M1 and proceed to M2. If it fails, retain the exact diagnostics and keep the live gate open; distinguish implementation defects from provider/configuration limitations. diff --git a/test/core/probes/native_codex_smoke.py b/test/core/probes/native_codex_smoke.py new file mode 100644 index 0000000..a3802bc --- /dev/null +++ b/test/core/probes/native_codex_smoke.py @@ -0,0 +1,189 @@ +#!/usr/bin/env python3 +"""Opt-in, bounded live smoke for the production Codex native M1 path.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +import time +from typing import Any + +CORE_SRC = Path(__file__).resolve().parents[3] / "plugin" / "core" / "src" +sys.path.insert(0, str(CORE_SRC)) + +from devsquad.adapters import AdapterManifest, harness_version, prepare_native_codex, prepare_native_codex_from_catalog +from devsquad.catalog import update_last_good +from devsquad.codex_protocol import ( + JsonLinePeer, + NativeTurnState, + discover_models, + initialize_request, + initialized_notification, + receive_response, + thread_start_request, + turn_start_request, +) + + +class RecordingPeer: + def __init__(self, peer: JsonLinePeer, transcript: Any): + self.peer, self.transcript = peer, transcript + + def send(self, message: dict[str, Any]) -> None: + self.transcript.write(json.dumps({"direction": "send", "message": message}) + "\n") + self.transcript.flush() + self.peer.send(message) + + def receive(self, timeout_seconds: float) -> dict[str, Any]: + message = self.peer.receive(timeout_seconds) + self.transcript.write(json.dumps({"direction": "receive", "message": message}) + "\n") + self.transcript.flush() + return message + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(65536), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _stop(process: subprocess.Popen[str]) -> str: + if process.poll() is None: + process.terminate() + try: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=2) + return f"exit:{process.returncode}" + + +def _server(spec: Any, stderr_path: Path, transcript_path: Path): + stderr_stream = stderr_path.open("w", encoding="utf-8") + transcript = transcript_path.open("w", encoding="utf-8") + environment = os.environ.copy() + environment.update(spec.env) + process = subprocess.Popen( + list(spec.argv), cwd=spec.cwd, env=environment, stdin=subprocess.PIPE, + stdout=subprocess.PIPE, stderr=stderr_stream, text=True, bufsize=1, + ) + assert process.stdin is not None and process.stdout is not None + peer = RecordingPeer(JsonLinePeer(process.stdout, process.stdin), transcript) + return process, peer, stderr_stream, transcript + + +def _initialize(peer: RecordingPeer, timeout: float) -> None: + peer.send(initialize_request(1)) + response = receive_response(peer, 1, timeout_seconds=timeout) + if "error" in response: + raise RuntimeError(f"initialize failed: {response['error']}") + peer.send(initialized_notification()) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--run-live", action="store_true", help="required acknowledgement for a live subscription-backed probe") + parser.add_argument("--output-dir", type=Path, default=Path.home() / ".devsquad" / "private-probes") + parser.add_argument("--timeout", type=int, default=30) + args = parser.parse_args() + if not args.run_live: + parser.error("--run-live is required") + if args.timeout < 5 or args.timeout > 60: + parser.error("--timeout must be between 5 and 60 seconds") + + stamp = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + revision = subprocess.run(["git", "rev-parse", "HEAD"], cwd=CORE_SRC, text=True, capture_output=True, check=True).stdout.strip() + run_dir = args.output_dir.expanduser().resolve() / f"native-codex-{stamp}-{revision[:12]}" + run_dir.mkdir(parents=True, mode=0o700, exist_ok=False) + os.chmod(run_dir, 0o700) + receipt: dict[str, Any] = {"started_at": stamp, "revision": revision, "status": "failed", "cleanup": []} + manifest_path = CORE_SRC.parents[1] / "adapters" / "codex" / "adapter.json" + manifest = AdapterManifest.load(manifest_path) + binary = manifest.resolve_binary() + if not binary: + raise RuntimeError("codex executable is unavailable") + version = harness_version(binary) + if not version: + raise RuntimeError("could not determine codex version") + + try: + with tempfile.TemporaryDirectory(prefix="devsquad-native-smoke-") as workspace_name: + workspace = Path(workspace_name) + subprocess.run(["git", "init", "-q", str(workspace)], check=True) + bootstrap = prepare_native_codex( + manifest.with_model_efforts({"bootstrap": ("low",)}), cwd=str(workspace), model="bootstrap", + effort="low", permission="read_only", timeout_seconds=args.timeout, harness_version_value=version, + ) + first, peer, stderr_stream, transcript = _server(bootstrap, run_dir / "discovery.stderr.log", run_dir / "discovery.jsonl") + try: + _initialize(peer, args.timeout) + models = discover_models(peer, first_request_id=10, timeout_seconds=args.timeout) + finally: + receipt["cleanup"].append({"discovery": _stop(first)}) + transcript.close(); stderr_stream.close() + + snapshot = update_last_good(run_dir / "catalog.json", harness="codex", version=version, models=models, complete=True) + candidates = [m for m in snapshot["models"] if m.get("supported_efforts")] + if not candidates: + raise RuntimeError("discovery returned no model with verified effort metadata") + selected = next((m for m in candidates if m.get("is_default")), candidates[0]) + supported = selected["supported_efforts"] + effort = next((name for name in ("minimal", "low", "medium", "high", "xhigh") if name in supported), supported[0]) + spec = prepare_native_codex_from_catalog( + manifest, snapshot, cwd=str(workspace), model=selected["id"], effort=effort, + permission="read_only", timeout_seconds=args.timeout, harness_version_value=version, + ) + second, peer, stderr_stream, transcript = _server(spec, run_dir / "turn.stderr.log", run_dir / "turn.jsonl") + try: + _initialize(peer, args.timeout) + peer.send(thread_start_request(20, cwd=str(workspace), model=selected["id"], permission="read_only")) + thread_response = receive_response(peer, 20, timeout_seconds=args.timeout) + result = thread_response.get("result", {}) + thread = result.get("thread", {}) if isinstance(result, dict) else {} + thread_id = thread.get("id") or result.get("threadId") + if not isinstance(thread_id, str) or not thread_id: + raise RuntimeError(f"thread/start returned no thread id: {thread_response}") + peer.send(turn_start_request(21, thread_id=thread_id, prompt="Reply with exactly DEVSQUAD_M1_NATIVE_OK. Do not use tools.", model=selected["id"], effort=effort, cwd=str(workspace), permission="read_only")) + turn_response = receive_response(peer, 21, timeout_seconds=args.timeout) + turn_result = turn_response.get("result", {}) + turn = turn_result.get("turn", {}) if isinstance(turn_result, dict) else {} + turn_id = turn.get("id") or turn_result.get("turnId") + if not isinstance(turn_id, str) or not turn_id: + raise RuntimeError(f"turn/start returned no turn id: {turn_response}") + state = NativeTurnState(thread_id=thread_id, turn_id=turn_id) + deadline = time.monotonic() + args.timeout + while not state.terminal: + state.consume(peer.receive(max(0.01, deadline - time.monotonic()))) + output = "".join(state.output).strip() + if state.terminal_status != "completed" or output != "DEVSQUAD_M1_NATIVE_OK": + raise RuntimeError(f"native verdict was not successful: status={state.terminal_status!r}, output={output!r}") + receipt.update({"status": "passed", "harness_version": version, "model": selected["id"], "effort": effort, "permission": "read_only", "terminal_status": state.terminal_status}) + finally: + receipt["cleanup"].append({"turn": _stop(second)}) + transcript.close(); stderr_stream.close() + except Exception as exc: + receipt["error_type"] = type(exc).__name__ + receipt["error"] = str(exc) + finally: + receipt["finished_at"] = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + for path in run_dir.iterdir(): + if path.is_file(): os.chmod(path, 0o600) + receipt["private_artifacts"] = {p.name: _sha256(p) for p in sorted(run_dir.iterdir()) if p.is_file()} + receipt_path = run_dir / "receipt.json" + receipt_path.write_text(json.dumps(receipt, indent=2, sort_keys=True) + "\n") + os.chmod(receipt_path, 0o600) + print(json.dumps({"status": receipt["status"], "run_dir": str(run_dir), "receipt_sha256": _sha256(receipt_path)})) + return 0 if receipt["status"] == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) From e6b9752ab9fbf4832510ae9e8f29579a07a93195 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:20:40 +0530 Subject: [PATCH 015/197] fix: resolve native probe manifest --- test/core/probes/native_codex_smoke.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/test/core/probes/native_codex_smoke.py b/test/core/probes/native_codex_smoke.py index a3802bc..a823122 100644 --- a/test/core/probes/native_codex_smoke.py +++ b/test/core/probes/native_codex_smoke.py @@ -106,7 +106,7 @@ def main() -> int: run_dir.mkdir(parents=True, mode=0o700, exist_ok=False) os.chmod(run_dir, 0o700) receipt: dict[str, Any] = {"started_at": stamp, "revision": revision, "status": "failed", "cleanup": []} - manifest_path = CORE_SRC.parents[1] / "adapters" / "codex" / "adapter.json" + manifest_path = CORE_SRC.parent / "adapters" / "codex" / "adapter.json" manifest = AdapterManifest.load(manifest_path) binary = manifest.resolve_binary() if not binary: From 5a373eb0714099b03973443aa8bd86d74f9650ec Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:21:27 +0530 Subject: [PATCH 016/197] fix: apply native probe environment --- test/core/probes/native_codex_smoke.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/test/core/probes/native_codex_smoke.py b/test/core/probes/native_codex_smoke.py index a823122..31cfc06 100644 --- a/test/core/probes/native_codex_smoke.py +++ b/test/core/probes/native_codex_smoke.py @@ -71,7 +71,7 @@ def _server(spec: Any, stderr_path: Path, transcript_path: Path): stderr_stream = stderr_path.open("w", encoding="utf-8") transcript = transcript_path.open("w", encoding="utf-8") environment = os.environ.copy() - environment.update(spec.env) + environment.update(spec.environment) process = subprocess.Popen( list(spec.argv), cwd=spec.cwd, env=environment, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=stderr_stream, text=True, bufsize=1, From 314c8ecc70a0dded7c9b4d3640019f9884deaeea Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:22:32 +0530 Subject: [PATCH 017/197] fix: bound native probe process ownership --- test/core/probes/native_codex_smoke.py | 37 +++++++++++++++++++------- 1 file changed, 28 insertions(+), 9 deletions(-) diff --git a/test/core/probes/native_codex_smoke.py b/test/core/probes/native_codex_smoke.py index 31cfc06..21f06b3 100644 --- a/test/core/probes/native_codex_smoke.py +++ b/test/core/probes/native_codex_smoke.py @@ -4,6 +4,7 @@ from __future__ import annotations import argparse +import errno import hashlib import json import os @@ -18,8 +19,9 @@ CORE_SRC = Path(__file__).resolve().parents[3] / "plugin" / "core" / "src" sys.path.insert(0, str(CORE_SRC)) -from devsquad.adapters import AdapterManifest, harness_version, prepare_native_codex, prepare_native_codex_from_catalog +from devsquad.adapters import AdapterManifest, harness_version, prepare_native_codex_from_catalog from devsquad.catalog import update_last_good +from devsquad.contracts import ExecutionIdentity, LaunchSpec, SCHEMA_VERSION from devsquad.codex_protocol import ( JsonLinePeer, NativeTurnState, @@ -56,15 +58,25 @@ def _sha256(path: Path) -> str: return digest.hexdigest() -def _stop(process: subprocess.Popen[str]) -> str: +def _stop(process: subprocess.Popen[str]) -> dict[str, Any]: + pgid = os.getpgid(process.pid) if process.poll() is None else process.pid if process.poll() is None: - process.terminate() + os.killpg(pgid, 15) try: process.wait(timeout=2) except subprocess.TimeoutExpired: - process.kill() + os.killpg(pgid, 9) process.wait(timeout=2) - return f"exit:{process.returncode}" + try: + os.killpg(pgid, 0) + except OSError as exc: + if exc.errno != errno.ESRCH: + raise + absent = True + else: + absent = False + os.killpg(pgid, 9) + return {"exit_code": process.returncode, "process_group": pgid, "group_absent": absent} def _server(spec: Any, stderr_path: Path, transcript_path: Path): @@ -75,6 +87,7 @@ def _server(spec: Any, stderr_path: Path, transcript_path: Path): process = subprocess.Popen( list(spec.argv), cwd=spec.cwd, env=environment, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=stderr_stream, text=True, bufsize=1, + start_new_session=True, ) assert process.stdin is not None and process.stdout is not None peer = RecordingPeer(JsonLinePeer(process.stdout, process.stdin), transcript) @@ -119,9 +132,12 @@ def main() -> int: with tempfile.TemporaryDirectory(prefix="devsquad-native-smoke-") as workspace_name: workspace = Path(workspace_name) subprocess.run(["git", "init", "-q", str(workspace)], check=True) - bootstrap = prepare_native_codex( - manifest.with_model_efforts({"bootstrap": ("low",)}), cwd=str(workspace), model="bootstrap", - effort="low", permission="read_only", timeout_seconds=args.timeout, harness_version_value=version, + bootstrap = LaunchSpec( + SCHEMA_VERSION, "codex", "native_protocol", + (binary, "app-server", "--listen", "stdio://"), str(workspace), None, + args.timeout, + ExecutionIdentity("codex", version, "openai", None, None, None, (), "read_only", None, "verified"), + {"DEVSQUAD_WORKER": "1"}, ) first, peer, stderr_stream, transcript = _server(bootstrap, run_dir / "discovery.stderr.log", run_dir / "discovery.jsonl") try: @@ -162,7 +178,10 @@ def main() -> int: state = NativeTurnState(thread_id=thread_id, turn_id=turn_id) deadline = time.monotonic() + args.timeout while not state.terminal: - state.consume(peer.receive(max(0.01, deadline - time.monotonic()))) + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("native turn did not reach correlated terminal state") + state.consume(peer.receive(remaining)) output = "".join(state.output).strip() if state.terminal_status != "completed" or output != "DEVSQUAD_M1_NATIVE_OK": raise RuntimeError(f"native verdict was not successful: status={state.terminal_status!r}, output={output!r}") From 97a10f0ae1e81e69557589492176dc5e3e54415d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:24:11 +0530 Subject: [PATCH 018/197] fix: enforce native event correlation --- docs/plans/engineering-team/RESUME.md | 18 +++++++-- plugin/core/src/devsquad/codex_protocol.py | 32 +++++++++++++--- test/core/probes/native_codex_smoke.py | 5 ++- test/core/test_m1_gate_review.py | 43 +++++++++++++++++++++- 4 files changed, 88 insertions(+), 10 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 32aa57f..26379c8 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -9,7 +9,9 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne - Last verified implementation checkpoint: `a67ab58`. - Last evidence/status checkpoint before this recovery note: `cc98ce7`. - Recovery checkpoint: `ad46b2f`. The opt-in saved probe is - `test/core/probes/native_codex_smoke.py`; commit it before its first live run. + `test/core/probes/native_codex_smoke.py`. Probe checkpoints `aa3fe1c`, + `e6b9752`, `5a373eb` and `314c8ec` preserve the script and bounded process + ownership repairs; the next checkpoint adds strict native event correlation. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 has not started. @@ -24,18 +26,28 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 32 tests passed | +| Python core discovery | 34 tests passed | | Bash 3.2 regression suite | 10 test files, 202 assertions passed | | Wheel installation | Temporary venv resolves packaged schemas, adapters and shared taxonomy | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | | Integrated native adapter proof | **Still pending** | +The first two saved-probe invocations failed before `Popen` because of +probe-only path/field defects, so neither launched Codex nor consumed a model +turn. Their private receipts remain under `~/.devsquad/private-probes`. The +probe now uses a dedicated process session, bounded group TERM/KILL cleanup, +an explicit terminal deadline, and retains early notifications for correlation. +After the latest native-state reviewer regressions, core discovery contains 34 +passing tests; update the authoritative evidence count with the eventual live +result. + The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed evidence is [M1-invocation-core-2026-09-06.json](evidence/M1-invocation-core-2026-09-06.json). [backlog.json](backlog.json) retains M1 as `in_progress`. Grok workspace-write and unprobed Antigravity/Grok settings are not advertised as verified. ## Exact next work 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. -2. Run the committed opt-in probe once with `--run-live`. It writes stdout +2. After committing the current protocol/probe/reviewer regression checkpoint, + run the opt-in probe once with `--run-live`. It writes stdout protocol frames and stderr to separate private files under `~/.devsquad/private-probes`, uses bounded deadlines, and records cleanup and hashes in its private receipt. Previous inline probes have no raw logs. diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index 6439f34..49552a6 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -125,12 +125,30 @@ def consume(self, message: dict[str, Any]) -> None: raise ContractError("native message must be an object") self.events.append(message) method = message.get("method", "") - params = message.get("params") or message.get("result") or {} + params = message.get("params", message.get("result", {})) + if method in {"thread/started", "thread/start/completed", "turn/started", "item/agentMessage/delta", "turn/output/delta", "turn/completed", "error"} and not isinstance(params, dict): + raise ContractError("native event params must be an object") + if not isinstance(params, dict): + return if method in {"thread/started", "thread/start/completed"}: - self.thread_id = params.get("thread", {}).get("id") or params.get("threadId") or self.thread_id + thread = params.get("thread", {}) + if not isinstance(thread, dict): + raise ContractError("native thread must be an object") + candidate = thread.get("id") or params.get("threadId") + if candidate is not None and not isinstance(candidate, str): + raise ContractError("native thread id must be a string") + if self.thread_id and candidate and candidate != self.thread_id: + return + self.thread_id = candidate or self.thread_id message_thread = params.get("threadId") turn = params.get("turn") or {} + if not isinstance(turn, dict): + raise ContractError("native turn must be an object") message_turn = turn.get("id") or params.get("turnId") + if message_thread is not None and not isinstance(message_thread, str): + raise ContractError("native thread id must be a string") + if message_turn is not None and not isinstance(message_turn, str): + raise ContractError("native turn id must be a string") if self.thread_id and message_thread and message_thread != self.thread_id: return if self.turn_id and message_turn and message_turn != self.turn_id: @@ -139,11 +157,15 @@ def consume(self, message: dict[str, Any]) -> None: self.thread_id = message_thread or self.thread_id self.turn_id = message_turn or self.turn_id if method in {"item/agentMessage/delta", "turn/output/delta"}: - self.output.append(params.get("delta", "")) - if method == "turn/completed" and self.turn_id and message_turn == self.turn_id: + delta = params.get("delta", "") + if not isinstance(delta, str): + raise ContractError("native output delta must be a string") + if self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: + self.output.append(delta) + if method == "turn/completed" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: self.terminal = True self.terminal_status = turn.get("status") - if method == "error" and self.turn_id and message_turn == self.turn_id and not params.get("willRetry", False): + if method == "error" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id and not params.get("willRetry", False): self.terminal = True self.terminal_status = "failed" self.error = params.get("error") diff --git a/test/core/probes/native_codex_smoke.py b/test/core/probes/native_codex_smoke.py index 21f06b3..3b07234 100644 --- a/test/core/probes/native_codex_smoke.py +++ b/test/core/probes/native_codex_smoke.py @@ -169,13 +169,16 @@ def main() -> int: if not isinstance(thread_id, str) or not thread_id: raise RuntimeError(f"thread/start returned no thread id: {thread_response}") peer.send(turn_start_request(21, thread_id=thread_id, prompt="Reply with exactly DEVSQUAD_M1_NATIVE_OK. Do not use tools.", model=selected["id"], effort=effort, cwd=str(workspace), permission="read_only")) - turn_response = receive_response(peer, 21, timeout_seconds=args.timeout) + early_notifications: list[dict[str, Any]] = [] + turn_response = receive_response(peer, 21, timeout_seconds=args.timeout, on_notification=early_notifications.append) turn_result = turn_response.get("result", {}) turn = turn_result.get("turn", {}) if isinstance(turn_result, dict) else {} turn_id = turn.get("id") or turn_result.get("turnId") if not isinstance(turn_id, str) or not turn_id: raise RuntimeError(f"turn/start returned no turn id: {turn_response}") state = NativeTurnState(thread_id=thread_id, turn_id=turn_id) + for notification in early_notifications: + state.consume(notification) deadline = time.monotonic() + args.timeout while not state.terminal: remaining = deadline - time.monotonic() diff --git a/test/core/test_m1_gate_review.py b/test/core/test_m1_gate_review.py index db155ff..de2eb0a 100644 --- a/test/core/test_m1_gate_review.py +++ b/test/core/test_m1_gate_review.py @@ -18,7 +18,7 @@ sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) from devsquad.adapters import classify_cli -from devsquad.codex_protocol import JsonLinePeer +from devsquad.codex_protocol import JsonLinePeer, NativeTurnState from devsquad.contracts import ( ContractError, ExecutionIdentity, LaunchSpec, validate_launch_payload, ) @@ -97,6 +97,47 @@ def test_partial_native_message_is_not_completion(self): class NativeFramingReview(unittest.TestCase): + def test_native_state_rejects_malformed_notification_values(self): + cases = [ + {"method": "turn/completed", "params": 42}, + {"method": "turn/completed", "params": {"turn": "invalid"}}, + {"method": "item/agentMessage/delta", "params": { + "threadId": "expected-thread", "turnId": "expected-turn", "delta": 42, + }}, + ] + for message in cases: + with self.subTest(message=message): + state = NativeTurnState(thread_id="expected-thread", turn_id="expected-turn") + with self.assertRaises(ContractError): + state.consume(message) + + def test_native_state_keeps_bound_identity_and_requires_correlation(self): + state = NativeTurnState(thread_id="expected-thread", turn_id="expected-turn") + for message in [ + {"method": "thread/started", "params": {"thread": {"id": "unrelated"}}}, + {"method": "item/agentMessage/delta", "params": {"delta": "no identities"}}, + {"method": "item/agentMessage/delta", "params": { + "threadId": "unrelated", "turnId": "expected-turn", "delta": "wrong thread", + }}, + {"method": "turn/completed", "params": { + "turn": {"id": "expected-turn", "status": "completed"}, + }}, + ]: + state.consume(message) + self.assertEqual(state.thread_id, "expected-thread") + self.assertEqual(state.turn_id, "expected-turn") + self.assertEqual(state.output, []) + self.assertFalse(state.terminal) + state.consume({"method": "item/agentMessage/delta", "params": { + "threadId": "expected-thread", "turnId": "expected-turn", "delta": "valid", + }}) + state.consume({"method": "turn/completed", "params": { + "threadId": "expected-thread", "turn": {"id": "expected-turn", "status": "completed"}, + }}) + self.assertEqual(state.output, ["valid"]) + self.assertEqual(state.terminal_status, "completed") + self.assertTrue(state.terminal) + def test_two_frames_in_one_write_are_both_available(self): read_fd, write_fd = os.pipe() with os.fdopen(read_fd, "r") as reader, os.fdopen(write_fd, "w") as writer: From 25eabe7a0ee14a8ba004ec0af4ebf26905aca4ea Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:24:29 +0530 Subject: [PATCH 019/197] docs: make interrupted build recovery discoverable --- AGENTS.md | 31 +++++++++++++++++++++++ README.md | 6 +++++ docs/plans/engineering-team/START-HERE.md | 14 +++++++--- 3 files changed, 47 insertions(+), 4 deletions(-) create mode 100644 AGENTS.md diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..76a3c53 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,31 @@ +# Working on DevSquad + +Read [CONTRIBUTING.md](CONTRIBUTING.md) before changing code. Run +`bash test/run.sh` before every commit; core changes also need the relevant +offline Python tests. Preserve the existing Bash 3.2 and optional-jq contracts. + +## Continuing the engineering-team build + +This build is already underway. After a usage limit, interrupted session or +handoff, read these files before starting work: + +1. [RESUME.md](docs/plans/engineering-team/RESUME.md): latest checkpoint, + verified results, open findings and exact next action. +2. [backlog.json](docs/plans/engineering-team/backlog.json): milestone status + and evidence. +3. [SOL-HANDOFF.md](docs/plans/engineering-team/SOL-HANDOFF.md): the full + authorized delivery scope and acceptance criteria. + +Compare the recovery note with `git status` and recent commits; retain work +newer than the note. Continue on `codex/engineering-team` unless the user +directs otherwise. Do not restart the architecture exercise or discard +implementation to return to `main`. + +Checkpoint coherent partial work and update RESUME.md before long probes or +handoffs. Record failures and incomplete gates truthfully. Finish a pause with +a clean tree; never use a stash as the recovery mechanism. Keep raw provider +diagnostics and credentials outside tracked evidence. + +Usage limits do not authorize buying credits, redeeming reset credits, using +a paid API fallback or changing global AI settings. The saved artifacts must +allow the next session to resume without relying on conversation memory. diff --git a/README.md b/README.md index 68169e8..ca824d4 100644 --- a/README.md +++ b/README.md @@ -22,6 +22,12 @@ --- +**Engineering-team build in progress:** continuing after an interrupted AI +session? Read the [recovery checkpoint](docs/plans/engineering-team/RESUME.md) +for saved work, verified results and the next action. The product described +below is the existing plugin; the new runner's status is tracked in the +[implementation plan](docs/plans/engineering-team/START-HERE.md). + ## The 30-second version You told Claude to delegate the boring stuff. It nodded. Then it read 40 files itself, blew through its context window, and you paid for every token. diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index fe221e6..a04cb7c 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -1,6 +1,9 @@ # DevSquad: coding-agent entry point -**Build status: planned, not implemented.** This packet follows the local/GitHub review and Dikshant's September 6 brief. Start with M1; do not run another open-ended architecture exercise. +**Build status: implementation in progress.** After an interruption, read +[RESUME.md](RESUME.md) first and compare it with current Git state. M1 has code +and passing offline evidence; its remaining gates are recorded in +[M1-STATUS.md](M1-STATUS.md). Do not restart the architecture exercise. **Full-build assignment:** Use [SOL-HANDOFF.md](SOL-HANDOFF.md) for the user's request to have Sol execute everything, test thoroughly and make normal use simple. It includes M1–M7 plus the opt-in Council feature, and adds guided task entry over the same contracts. @@ -23,7 +26,7 @@ flowchart LR 1. Read [ADR-002](../../adr/ADR-002-surface-independent-engineering-team.md) for boundaries and decisions. 2. Implement the [contracts](CONTRACTS.md), using the [examples](examples/branch-review.json) as fixtures, not live model configuration. -3. Work through [IMPLEMENTATION](IMPLEMENTATION.md), one milestone at a time. [backlog.json](backlog.json) is the completion record; all milestones initially have `status: pending` and empty evidence. +3. Work through [IMPLEMENTATION](IMPLEMENTATION.md), one milestone at a time. [backlog.json](backlog.json) is the current completion record; resume the earliest unfinished requirement whose dependencies are ready. 4. Consult the [assessment](../../audits/2026-09-06-engineering-team-assessment.md) for verified defects and history, and [ADR-001](../../adr/ADR-001-contract-and-ledger-core.md) for legacy constraints retained by ADR-002. Selection is automatic by default, with validated per-role profile overrides. Read the [selection and LLM Council amendment](SELECTION-AND-COUNCIL.md) for the clarified contract. Its optional C1 extension follows M6 and does not block the seven core milestones. @@ -34,7 +37,8 @@ Also read the [native adapters and model lifecycle amendment](MODEL-LIFECYCLE-AN ```text Implement DevSquad's September engineering-team plan in this repository. -Read docs/plans/engineering-team/START-HERE.md and its contracts first. +Read docs/plans/engineering-team/RESUME.md, current Git state, +START-HERE.md and its contracts first. Preserve existing implementation. Start at the earliest pending milestone whose dependencies are complete. Implement M1 and pass its gate, then continue through M2–M7 and C1. Preserve existing Bash 3.2 wrapper callers and their four error prefixes. @@ -68,4 +72,6 @@ First usable product: **a saved branch review**. Next: **one bounded code change Defer a dashboard, universal DAG builder, remote execution service, automatic model training/router, plugin marketplace, autonomous merges and scheduled documentation jobs. Existing plugin behavior remains available while the new runner is opt-in; switching hook suggestions to the new route source happens only after its gate passes. -No new runtime, host registration, provider call, deployment or automation was created by this architecture packet. +This packet began as architecture only. Current implementation and live-probe +evidence are tracked in RESUME.md, the milestone status and backlog; they do +not yet establish that the full engineering-team product is usable. From 7b5c41c7d249962d729a91c65e4a084285820eb9 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:26:17 +0530 Subject: [PATCH 020/197] fix: confirm native probe group cleanup --- test/core/probes/native_codex_smoke.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/test/core/probes/native_codex_smoke.py b/test/core/probes/native_codex_smoke.py index 3b07234..bf30809 100644 --- a/test/core/probes/native_codex_smoke.py +++ b/test/core/probes/native_codex_smoke.py @@ -76,6 +76,16 @@ def _stop(process: subprocess.Popen[str]) -> dict[str, Any]: else: absent = False os.killpg(pgid, 9) + deadline = time.monotonic() + 2 + while time.monotonic() < deadline: + try: + os.killpg(pgid, 0) + except OSError as exc: + if exc.errno != errno.ESRCH: + raise + absent = True + break + time.sleep(0.05) return {"exit_code": process.returncode, "process_group": pgid, "group_absent": absent} From c50c6b423efcce9d3b3d63922ca999bbafc72d5e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:27:19 +0530 Subject: [PATCH 021/197] docs: record accepted M1 gate --- docs/plans/engineering-team/M1-STATUS.md | 12 +++++---- docs/plans/engineering-team/RESUME.md | 25 +++++++++++++------ docs/plans/engineering-team/backlog.json | 12 ++++----- .../M1-invocation-core-2026-09-06.json | 23 +++++++++-------- 4 files changed, 43 insertions(+), 29 deletions(-) diff --git a/docs/plans/engineering-team/M1-STATUS.md b/docs/plans/engineering-team/M1-STATUS.md index 41c5318..2be580f 100644 --- a/docs/plans/engineering-team/M1-STATUS.md +++ b/docs/plans/engineering-team/M1-STATUS.md @@ -1,8 +1,10 @@ # M1 implementation status -M1 is implemented through candidate `a67ab58` and remains **in progress** -pending independent review and a successful integrated native live probe. M2 -process ownership has not started. +M1 is **complete** for its bounded invocation and preparation scope at +`97a10f0`. Independent review accepted it after the 34-test core suite, the +202-assertion shell suite, and inspection of the saved native receipt. Probe +cleanup hardening is preserved at `7b5c41c`. M2 process ownership has not +started. | # | Requirement | Evidence | Status | |---|---|---|---| @@ -18,8 +20,8 @@ process ownership has not started. | 10 | Tracked, bounded Antigravity context | Literal Git inventory; ignored/binary/secret/oversize omissions; file and ancestor-symlink containment tests | verified offline | | 11 | Package and CLI envelope | Temporary wheel installation resolves schemas, adapters and shared taxonomy; input/readiness exit behavior is tested | verified offline | | 12 | Bounded real starting-profile smoke | Codex CLI `gpt-5.5`, low effort, read-only, ephemeral JSONL invocation completed on 2026-09-06 | verified live (CLI) | -| 13 | Integrated native preparation/protocol/classification probe | Correct handshake initializes, but the earlier inline app-server attempt stalled at `model/list` or `thread/start`; root is diagnosing with a saved probe | pending live | -| 14 | Independent gate review | Reviewer regressions are tracked and pass; final root disposition remains outstanding | pending final review | +| 13 | Integrated native preparation/protocol/classification probe | Saved probe at `97a10f0`: discovery → catalog → preparation → read-only gpt-5.5/low turn → correlated completion; private receipt and stream hashes retained | verified live | +| 14 | Independent gate review | Eight reviewer regressions pass; reviewer independently reran 34 core tests, inspected all private artifact hashes and verified requested gpt-5.5/low plus readOnly/networkAccess:false and correlated completion | verified | The Python bridge returns launch/protocol preparation and normalized parser policy only. It does not spawn a worker or implement a second watchdog. M2 diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 26379c8..003dae9 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -12,6 +12,13 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne `test/core/probes/native_codex_smoke.py`. Probe checkpoints `aa3fe1c`, `e6b9752`, `5a373eb` and `314c8ec` preserve the script and bounded process ownership repairs; the next checkpoint adds strict native event correlation. +- Native correlation/probe checkpoint: `97a10f0`. The saved integrated probe + passed; its private receipt SHA256 is + `324c2ce6154936ecf71a8bf913a195befd4251efc6dbb8bbab3dc0dbe1b86df8`. + Exact private receipt directory: + `/Users/Dikshant/.devsquad/private-probes/native-codex-20260907T025419Z-97a10f0ae1e8`. +- M1 is accepted for its bounded invocation/preparation scope. Probe cleanup + polling is checkpointed at `7b5c41c`; the next milestone is M2. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 has not started. @@ -30,7 +37,7 @@ Verified at the implementation/evidence checkpoints above: | Bash 3.2 regression suite | 10 test files, 202 assertions passed | | Wheel installation | Temporary venv resolves packaged schemas, adapters and shared taxonomy | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | -| Integrated native adapter proof | **Still pending** | +| Integrated native adapter proof | Passed at `97a10f0`; gpt-5.5/low, read-only, correlated completion and confirmed process-group cleanup | The first two saved-probe invocations failed before `Popen` because of probe-only path/field defects, so neither launched Codex nor consumed a model @@ -39,19 +46,21 @@ probe now uses a dedicated process session, bounded group TERM/KILL cleanup, an explicit terminal deadline, and retains early notifications for correlation. After the latest native-state reviewer regressions, core discovery contains 34 passing tests; update the authoritative evidence count with the eventual live -result. +result. The successful run retained separate stderr files of 138,030 and +285,644 bytes, supporting the diagnosis that an undrained stderr pipe caused +the earlier apparent nonresponses. The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed evidence is [M1-invocation-core-2026-09-06.json](evidence/M1-invocation-core-2026-09-06.json). [backlog.json](backlog.json) retains M1 as `in_progress`. Grok workspace-write and unprobed Antigravity/Grok settings are not advertised as verified. ## Exact next work 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. -2. After committing the current protocol/probe/reviewer regression checkpoint, - run the opt-in probe once with `--run-live`. It writes stdout - protocol frames and stderr to separate private files under - `~/.devsquad/private-probes`, uses bounded deadlines, and records cleanup and - hashes in its private receipt. Previous inline probes have no raw logs. -3. The probe exercises the actual path: initialized app-server → complete model discovery → saved snapshot → prepared LaunchSpec → native thread/turn → correlated terminal result. It uses one bounded read-only inference, a temporary Git workspace and existing subscription authentication. +2. Review the saved integrated native result and the complete M1 matrix. The + opt-in probe writes protocol frames and stderr to separate private files + under `~/.devsquad/private-probes`, uses bounded deadlines, and records + cleanup and hashes in its private receipt. +3. Start only the bounded M2 assignment supplied by root. Preserve the M1 + limitations below; acceptance is not a claim that the full product exists. 4. Investigate the nonresponse before declaring an external blocker. Prior probes reported successful initialization but no `model/list` or `thread/start` reply after bounded waits, despite sending the required `initialized` notification. They reported an installed Codex 0.135.0 warning about the global `ultra` effort value. Per-process overrides were attempted; global settings were not changed. **A new hypothesis to test is blocked stderr output from an undrained subprocess PIPE.** Redirect stderr to a private file or drain it concurrently; a full stderr pipe can stall a child. This cause is not yet established. 5. If the actual adapter path succeeds, record the exact revision/commands/outcome and complete the remaining independent M1 review. Only then close M1 and proceed to M2. If it fails, retain the exact diagnostics and keep the live gate open; distinguish implementation defects from provider/configuration limitations. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index eeb4c0b..68364ab 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -11,26 +11,26 @@ "execution_brief": "SOL-HANDOFF.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", - "next_milestone": "M1", + "next_milestone": "M2", "milestones": [ { "id": "M1", "title": "Truthful reusable adapter invocation", "depends_on": [], - "status": "in_progress", + "status": "complete", "acceptance_section": "M1 — Make invocation truthful and reusable", "evidence": [ { "kind": "implementation_checkpoint", - "revision": "a67ab58", - "command_or_action": "32 core tests, 202 shell assertions, temporary wheel install, prior live Codex metadata and read-only CLI smoke", - "outcome": "All enumerated offline M1 gates pass; saved integrated native live probe and independent final review remain open", + "revision": "97a10f0", + "command_or_action": "34 core tests, 202 shell assertions, temporary wheel install, and saved integrated native read-only smoke", + "outcome": "All enumerated offline and live M1 gates pass; independent review accepted the bounded invocation/preparation scope", "artifact": "evidence/M1-invocation-core-2026-09-06.json", "recorded_at": "2026-09-07T08:20:00+05:30", "availability": "portable_redacted" } ], - "blocker": "Integrated native app-server probe and independent final M1 review are pending." + "blocker": null }, { "id": "M2", diff --git a/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json b/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json index 83b87ea..2051734 100644 --- a/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json +++ b/docs/plans/engineering-team/evidence/M1-invocation-core-2026-09-06.json @@ -1,9 +1,10 @@ { "schema_version": 1, "milestone": "M1", - "status": "in_progress", - "implementation_revision": "a67ab58", - "recorded_at": "2026-09-07T08:20:00+05:30", + "status": "complete", + "implementation_revision": "97a10f0", + "verification_revision": "7b5c41c", + "recorded_at": "2026-09-07T08:30:00+05:30", "evidence": [ { "kind": "offline_test", @@ -14,7 +15,7 @@ { "kind": "offline_test", "command_or_action": "PYTHONDONTWRITEBYTECODE=1 python3 -m unittest discover -s test/core -v", - "outcome": "32 tests passed, including six independent reviewer regressions", + "outcome": "34 tests passed, including eight independent reviewer regressions", "availability": "tracked tests" }, { @@ -38,12 +39,14 @@ { "kind": "live_integrated_probe", "command_or_action": "production preparation and JSON-line peer: initialize response, initialized notification, then model/list or prepared thread/start; bounded per-process low-effort override", - "outcome": "Earlier inline attempt initialized, then did not receive model/list or thread/start within its bound. Root is rerunning a saved diagnostic probe, including stderr draining; no integrated success is claimed yet", - "availability": "pending saved diagnostic evidence" + "outcome": "Passed at 97a10f0: complete discovery selected gpt-5.5/low, catalog-backed preparation launched a fresh server, and the read-only no-tools turn returned the exact expected output with correlated terminal completion. Both owned process groups were confirmed absent after cleanup. Receipt SHA256: 324c2ce6154936ecf71a8bf913a195befd4251efc6dbb8bbab3dc0dbe1b86df8", + "availability": "private receipt and stream hashes retained locally; redacted summary only" } ], - "residual_blockers": [ - "The integrated native app-server path has not yet completed a saved real turn probe.", - "Independent final M1 gate review is pending." - ] + "independent_review": { + "status": "accepted", + "scope": "bounded M1 invocation and preparation", + "outcome": "Reviewer reran 34 core tests, relied on the required 202-assertion shell run, verified the receipt and all five private artifact hashes, and inspected requested gpt-5.5/low, readOnly with networkAccess false, and correlated turn completion." + }, + "residual_blockers": [] } From 77cbed35c46bb13022e639d487e12e9970b442c7 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 08:28:22 +0530 Subject: [PATCH 022/197] docs: make M2 the recovery next step --- docs/plans/engineering-team/RESUME.md | 34 +++++++++++---------------- 1 file changed, 14 insertions(+), 20 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 003dae9..990fc2d 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,19 +6,16 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` is the published runtime baseline. -- Last verified implementation checkpoint: `a67ab58`. -- Last evidence/status checkpoint before this recovery note: `cc98ce7`. -- Recovery checkpoint: `ad46b2f`. The opt-in saved probe is - `test/core/probes/native_codex_smoke.py`. Probe checkpoints `aa3fe1c`, - `e6b9752`, `5a373eb` and `314c8ec` preserve the script and bounded process - ownership repairs; the next checkpoint adds strict native event correlation. +- Accepted M1 implementation checkpoint: `97a10f0`. +- Probe cleanup hardening checkpoint: `7b5c41c`. +- Accepted M1 status/evidence checkpoint: `c50c6b4`. - Native correlation/probe checkpoint: `97a10f0`. The saved integrated probe passed; its private receipt SHA256 is `324c2ce6154936ecf71a8bf913a195befd4251efc6dbb8bbab3dc0dbe1b86df8`. Exact private receipt directory: `/Users/Dikshant/.devsquad/private-probes/native-codex-20260907T025419Z-97a10f0ae1e8`. -- M1 is accepted for its bounded invocation/preparation scope. Probe cleanup - polling is checkpointed at `7b5c41c`; the next milestone is M2. +- M1 is accepted for its bounded invocation/preparation scope. The next + milestone is M2. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 has not started. @@ -44,25 +41,22 @@ probe-only path/field defects, so neither launched Codex nor consumed a model turn. Their private receipts remain under `~/.devsquad/private-probes`. The probe now uses a dedicated process session, bounded group TERM/KILL cleanup, an explicit terminal deadline, and retains early notifications for correlation. -After the latest native-state reviewer regressions, core discovery contains 34 -passing tests; update the authoritative evidence count with the eventual live -result. The successful run retained separate stderr files of 138,030 and +The successful run retained separate stderr files of 138,030 and 285,644 bytes, supporting the diagnosis that an undrained stderr pipe caused the earlier apparent nonresponses. -The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed evidence is [M1-invocation-core-2026-09-06.json](evidence/M1-invocation-core-2026-09-06.json). [backlog.json](backlog.json) retains M1 as `in_progress`. Grok workspace-write and unprobed Antigravity/Grok settings are not advertised as verified. +The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed evidence is [M1-invocation-core-2026-09-06.json](evidence/M1-invocation-core-2026-09-06.json). [backlog.json](backlog.json) marks M1 complete and M2 next. Grok workspace-write and unprobed Antigravity/Grok settings are not advertised as verified. ## Exact next work 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. -2. Review the saved integrated native result and the complete M1 matrix. The - opt-in probe writes protocol frames and stderr to separate private files - under `~/.devsquad/private-probes`, uses bounded deadlines, and records - cleanup and hashes in its private receipt. -3. Start only the bounded M2 assignment supplied by root. Preserve the M1 - limitations below; acceptance is not a claim that the full product exists. -4. Investigate the nonresponse before declaring an external blocker. Prior probes reported successful initialization but no `model/list` or `thread/start` reply after bounded waits, despite sending the required `initialized` notification. They reported an installed Codex 0.135.0 warning about the global `ultra` effort value. Per-process overrides were attempted; global settings were not changed. **A new hypothesis to test is blocked stderr output from an undrained subprocess PIPE.** Redirect stderr to a private file or drain it concurrently; a full stderr pipe can stall a child. This cause is not yet established. -5. If the actual adapter path succeeds, record the exact revision/commands/outcome and complete the remaining independent M1 review. Only then close M1 and proceed to M2. If it fails, retain the exact diagnostics and keep the live gate open; distinguish implementation defects from provider/configuration limitations. +2. Read the M2 section of [IMPLEMENTATION.md](IMPLEMENTATION.md) and the + corresponding contracts before editing. Implement the bounded M2 store + foundation assigned by root, then the supervisor/recovery slices in their + dependency order. +3. Preserve M1 limitations and process-ownership boundaries. M1 acceptance is + not a claim that the full product exists, and no additional native probe is + needed for the accepted gate. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. From 0c2929c2923a954f1bb9460ccf825a71d38d7732 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 13:22:57 +0530 Subject: [PATCH 023/197] feat: add transactional M2 store foundation --- plugin/core/pyproject.toml | 3 + .../src/devsquad/migrations/001_initial.sql | 57 ++++ .../core/src/devsquad/migrations/__init__.py | 1 + plugin/core/src/devsquad/store.py | 276 ++++++++++++++++++ test/core/test_m2_gate_review.py | 124 ++++++++ test/core/test_store.py | 121 ++++++++ 6 files changed, 582 insertions(+) create mode 100644 plugin/core/src/devsquad/migrations/001_initial.sql create mode 100644 plugin/core/src/devsquad/migrations/__init__.py create mode 100644 plugin/core/src/devsquad/store.py create mode 100644 test/core/test_m2_gate_review.py create mode 100644 test/core/test_store.py diff --git a/plugin/core/pyproject.toml b/plugin/core/pyproject.toml index 22b82fc..5cec46b 100644 --- a/plugin/core/pyproject.toml +++ b/plugin/core/pyproject.toml @@ -17,6 +17,9 @@ package-dir = {"" = "src"} [tool.setuptools.packages.find] where = ["src"] +[tool.setuptools.package-data] +"devsquad.migrations" = ["*.sql"] + [tool.setuptools.data-files] "share/devsquad/adapters/codex" = ["adapters/codex/adapter.json"] "share/devsquad/adapters/antigravity" = ["adapters/antigravity/adapter.json"] diff --git a/plugin/core/src/devsquad/migrations/001_initial.sql b/plugin/core/src/devsquad/migrations/001_initial.sql new file mode 100644 index 0000000..0dead90 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/001_initial.sql @@ -0,0 +1,57 @@ +CREATE TABLE schema_migrations ( + version INTEGER PRIMARY KEY, + applied_at TEXT NOT NULL +); + +CREATE TABLE projects ( + id TEXT PRIMARY KEY, + git_common_dir TEXT NOT NULL UNIQUE, + created_at TEXT NOT NULL +); + +CREATE TABLE runs ( + id TEXT PRIMARY KEY, + project_id TEXT NOT NULL REFERENCES projects(id), + idempotency_key TEXT NOT NULL, + request_hash TEXT NOT NULL, + submitted_request TEXT NOT NULL, + mutable_snapshot TEXT, + state TEXT NOT NULL, + phase TEXT, + version INTEGER NOT NULL, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + UNIQUE(project_id, idempotency_key) +); + +CREATE TABLE claims ( + run_id TEXT PRIMARY KEY REFERENCES runs(id), + kind TEXT NOT NULL, + fencing_token INTEGER NOT NULL, + owner_id TEXT NOT NULL, + active INTEGER NOT NULL CHECK(active IN (0, 1)), + claimed_at TEXT NOT NULL +); + +CREATE TABLE events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + run_id TEXT NOT NULL REFERENCES runs(id), + run_version INTEGER NOT NULL, + type TEXT NOT NULL, + payload TEXT NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(run_id, run_version) +); + +CREATE TABLE artifacts ( + id TEXT PRIMARY KEY, + run_id TEXT NOT NULL REFERENCES runs(id), + name TEXT NOT NULL, + path TEXT NOT NULL UNIQUE, + sha256 TEXT NOT NULL, + byte_size INTEGER NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(run_id, name) +); + +CREATE INDEX events_run_cursor ON events(run_id, id); diff --git a/plugin/core/src/devsquad/migrations/__init__.py b/plugin/core/src/devsquad/migrations/__init__.py new file mode 100644 index 0000000..690dbef --- /dev/null +++ b/plugin/core/src/devsquad/migrations/__init__.py @@ -0,0 +1 @@ +"""Packaged SQLite migrations for the local DevSquad ledger.""" diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py new file mode 100644 index 0000000..a56dcbd --- /dev/null +++ b/plugin/core/src/devsquad/store.py @@ -0,0 +1,276 @@ +"""Transactional SQLite ledger and atomic artifact storage for M2.""" + +from __future__ import annotations + +from dataclasses import dataclass +from datetime import datetime, timezone +import hashlib +from importlib.resources import files +import json +import os +from pathlib import Path +import sqlite3 +import subprocess +import tempfile +import uuid +from typing import Any + +from .contracts import ContractError + +SUPPORTED_SCHEMA_VERSION = 1 +TERMINAL_STATES = {"succeeded", "failed", "cancelled"} + + +class ConflictError(ContractError): + code = "CONFLICT" + + +class SchemaVersionError(ContractError): + code = "SCHEMA_UNSUPPORTED" + + +@dataclass(frozen=True) +class StartClaim: + run_id: str + project_id: str + request_hash: str + version: int + created: bool + fencing_token: int | None + + +def _utc_now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def canonical_json(value: Any) -> str: + try: + return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False) + except (TypeError, ValueError) as exc: + raise ContractError("request must be finite JSON") from exc + + +def request_hash(value: Any) -> str: + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def git_common_dir(worktree: Path) -> Path: + try: + result = subprocess.run( + ["git", "-C", str(worktree), "rev-parse", "--path-format=absolute", "--git-common-dir"], + text=True, capture_output=True, check=True, + ) + except (OSError, subprocess.CalledProcessError) as exc: + raise ContractError("project path is not a Git worktree") from exc + return Path(result.stdout.strip()).resolve(strict=True) + + +class Store: + def __init__(self, database: Path, artifacts: Path): + self.database = database + self.artifacts = artifacts + database.parent.mkdir(parents=True, exist_ok=True) + artifacts.mkdir(parents=True, exist_ok=True) + self.connection = sqlite3.connect(database, timeout=10, isolation_level=None) + self.connection.row_factory = sqlite3.Row + self.connection.execute("PRAGMA busy_timeout=10000") + self.connection.execute("PRAGMA foreign_keys=ON") + self.migrate() + self.connection.execute("PRAGMA journal_mode=WAL") + self.connection.execute("PRAGMA synchronous=FULL") + + def close(self) -> None: + self.connection.close() + + def migrate(self) -> None: + self.connection.execute("BEGIN EXCLUSIVE") + try: + table = self.connection.execute("SELECT 1 FROM sqlite_master WHERE type='table' AND name='schema_migrations'").fetchone() + current = self.connection.execute("SELECT COALESCE(MAX(version), 0) FROM schema_migrations").fetchone()[0] if table else 0 + if current > SUPPORTED_SCHEMA_VERSION: + raise SchemaVersionError(f"database schema {current} is newer than supported {SUPPORTED_SCHEMA_VERSION}") + if current == 0: + sql = files("devsquad.migrations").joinpath("001_initial.sql").read_text() + for statement in sql.split(";"): + if statement.strip(): + self.connection.execute(statement) + self.connection.execute("INSERT INTO schema_migrations(version, applied_at) VALUES(1, ?)", (_utc_now(),)) + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + + def _project(self, common_dir: Path) -> str: + key = str(common_dir) + row = self.connection.execute("SELECT id FROM projects WHERE git_common_dir=?", (key,)).fetchone() + if row: + return row[0] + project_id = str(uuid.uuid4()) + self.connection.execute("INSERT INTO projects(id, git_common_dir, created_at) VALUES(?,?,?)", (project_id, key, _utc_now())) + return project_id + + def claim_start(self, worktree: Path, idempotency_key: str, submitted_request: Any, owner_id: str) -> StartClaim: + if not idempotency_key or not owner_id: + raise ContractError("idempotency key and owner are required") + encoded, digest = canonical_json(submitted_request), request_hash(submitted_request) + common_dir = git_common_dir(worktree) + self.connection.execute("BEGIN IMMEDIATE") + try: + project_id = self._project(common_dir) + existing = self.connection.execute( + "SELECT id, request_hash, version FROM runs WHERE project_id=? AND idempotency_key=?", + (project_id, idempotency_key), + ).fetchone() + if existing: + if existing["request_hash"] != digest: + raise ConflictError("idempotency key was already used with a different request") + self.connection.execute("COMMIT") + return StartClaim(existing["id"], project_id, digest, existing["version"], False, None) + run_id, now = str(uuid.uuid4()), _utc_now() + self.connection.execute( + "INSERT INTO runs(id,project_id,idempotency_key,request_hash,submitted_request,state,phase,version,created_at,updated_at) VALUES(?,?,?,?,?,'queued','preparing',1,?,?)", + (run_id, project_id, idempotency_key, digest, encoded, now, now), + ) + self.connection.execute( + "INSERT INTO claims(run_id,kind,fencing_token,owner_id,active,claimed_at) VALUES(?,'preparing',1,?,1,?)", + (run_id, owner_id, now), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,1,'run.preparing','{}',?)", + (run_id, now), + ) + self.connection.execute("COMMIT") + return StartClaim(run_id, project_id, digest, 1, True, 1) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def complete_preparation(self, run_id: str, fencing_token: int, mutable_snapshot: Any) -> int: + snapshot = canonical_json(mutable_snapshot) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,c.fencing_token,c.active FROM runs r JOIN claims c ON c.run_id=r.id WHERE r.id=?", + (run_id,), + ).fetchone() + if not row or row["state"] != "queued" or row["phase"] != "preparing" or not row["active"] or row["fencing_token"] != fencing_token: + raise ConflictError("preparation claim is stale or cancelled") + version, now = row["version"] + 1, _utc_now() + self.connection.execute("UPDATE runs SET mutable_snapshot=?,phase=NULL,version=?,updated_at=? WHERE id=?", (snapshot, version, now, run_id)) + self.connection.execute("UPDATE claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.queued','{}',?)", (run_id, version, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def cancel_preparing(self, run_id: str) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT state,phase,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: + raise ContractError("run does not exist") + if row["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return row["version"] + if row["state"] != "queued" or row["phase"] != "preparing": + raise ConflictError("run is no longer preparing") + version, now = row["version"] + 1, _utc_now() + self.connection.execute("UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("UPDATE claims SET active=0,fencing_token=fencing_token+1 WHERE run_id=?", (run_id,)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelled','{}',?)", (run_id, version, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def append_event(self, run_id: str, expected_version: int, event_type: str, payload: Any, state: str | None = None) -> int: + if state is not None: + raise ConflictError("generic events cannot change run state") + encoded = canonical_json(payload) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT state,phase,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not row or row["version"] != expected_version: + raise ConflictError("run version changed") + if row["state"] in TERMINAL_STATES: + raise ConflictError("terminal run is immutable") + if row["phase"] is not None: + raise ConflictError("run phase is owned by a fenced operation") + version, now = expected_version + 1, _utc_now() + self.connection.execute("UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", (run_id, version, event_type, encoded, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def finalize_artifact(self, run_id: str, name: str, content: bytes) -> tuple[Path, str, int]: + if not name or Path(name).name != name: + raise ContractError("artifact name must be a single path component") + row = self.connection.execute("SELECT id FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: + raise ContractError("run does not exist") + digest = hashlib.sha256(content).hexdigest() + directory = self.artifacts / run_id + directory.mkdir(parents=True, exist_ok=True) + destination = directory / f"{digest}.blob" + descriptor, temporary = tempfile.mkstemp(prefix=f".{name}.", dir=directory) + try: + with os.fdopen(descriptor, "wb") as stream: + stream.write(content); stream.flush(); os.fsync(stream.fileno()) + try: + os.link(temporary, destination) + except FileExistsError: + if hashlib.sha256(destination.read_bytes()).hexdigest() != digest: + raise ConflictError("content-addressed artifact path is corrupt") + directory_fd = os.open(directory, os.O_RDONLY) + try: os.fsync(directory_fd) + finally: os.close(directory_fd) + finally: + if os.path.exists(temporary): os.unlink(temporary) + return destination, digest, len(content) + + def reference_artifact(self, run_id: str, name: str, path: Path, expected_sha256: str) -> str: + if not name or Path(name).name != name: + raise ContractError("artifact name must be a single path component") + expected_parent = (self.artifacts / run_id).resolve() + if path.resolve().parent != expected_parent: + raise ContractError("artifact path is outside the run-owned store") + content = path.read_bytes() + actual = hashlib.sha256(content).hexdigest() + if actual != expected_sha256: + raise ConflictError("artifact hash changed before database reference") + artifact_id = str(uuid.uuid4()) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: + raise ContractError("run does not exist") + if row["state"] in TERMINAL_STATES: + raise ConflictError("terminal run is immutable") + version, now = row["version"] + 1, _utc_now() + self.connection.execute("INSERT INTO artifacts(id,run_id,name,path,sha256,byte_size,created_at) VALUES(?,?,?,?,?,?,?)", (artifact_id, run_id, name, str(path), actual, len(content), _utc_now())) + self.connection.execute("UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id)) + event = canonical_json({"artifact_id": artifact_id, "name": name, "sha256": actual, "byte_size": len(content)}) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'artifact.recorded',?,?)", (run_id, version, event, now)) + self.connection.execute("COMMIT") + return artifact_id + except sqlite3.IntegrityError as exc: + self.connection.execute("ROLLBACK") + raise ConflictError("artifact name is already referenced for this run") from exc + except Exception: + self.connection.execute("ROLLBACK") + raise + + def store_artifact(self, run_id: str, name: str, content: bytes) -> str: + path, digest, _ = self.finalize_artifact(run_id, name, content) + return self.reference_artifact(run_id, name, path, digest) + + def run(self, run_id: str) -> dict[str, Any]: + row = self.connection.execute("SELECT * FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: raise ContractError("run does not exist") + return dict(row) diff --git a/test/core/test_m2_gate_review.py b/test/core/test_m2_gate_review.py new file mode 100644 index 0000000..1ce5470 --- /dev/null +++ b/test/core/test_m2_gate_review.py @@ -0,0 +1,124 @@ +"""Independent M2 checks for persisted ownership and artifact integrity.""" +from __future__ import annotations + +import json +import multiprocessing +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) + +from devsquad.contracts import ContractError +from devsquad.store import ConflictError, Store + + +def concurrent_open(database, artifacts, barrier, results): + store = None + try: + barrier.wait(timeout=10) + store = Store(Path(database), Path(artifacts)) + results.put(None) + except Exception as exc: + results.put(f"{type(exc).__name__}: {exc}") + finally: + if store is not None: + store.close() + + +class StoreIntegrityReview(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-store-review-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + self.store = Store(self.root / "state.sqlite3", self.root / "artifacts") + self.addCleanup(self.store.close) + self.task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + self.task["project"]["repo_path"] = str(self.repo) + self.claim = self.store.claim_start(self.repo, "review-key", self.task, "first-owner") + + def test_duplicate_artifact_never_changes_existing_receipt_content(self): + artifact_id = self.store.store_artifact(self.claim.run_id, "receipt.json", b"original") + row = self.store.connection.execute("SELECT path,sha256 FROM artifacts WHERE id=?", (artifact_id,)).fetchone() + path, original_digest = Path(row["path"]), row["sha256"] + try: + self.store.store_artifact(self.claim.run_id, "receipt.json", b"replacement") + except ContractError: + pass + self.assertEqual(path.read_bytes(), b"original") + row = self.store.connection.execute("SELECT sha256 FROM artifacts WHERE id=?", (artifact_id,)).fetchone() + self.assertEqual(row["sha256"], original_digest) + + def test_artifact_finalization_cannot_escape_with_run_id(self): + outside = self.root / "outside" + for run_id in (str(outside), "../outside"): + with self.subTest(run_id=run_id), self.assertRaises(ContractError): + self.store.finalize_artifact(run_id, "receipt.json", b"unowned") + self.assertFalse(outside.exists()) + + def test_queued_preparation_cannot_be_made_runnable_by_generic_event(self): + run = self.store.run(self.claim.run_id) + self.assertEqual(run["state"], "queued") + self.assertEqual(run["phase"], "preparing") + with self.assertRaises(ConflictError): + self.store.append_event(self.claim.run_id, run["version"], "run.started", {}, state="running") + self.assertEqual(self.store.run(self.claim.run_id)["phase"], "preparing") + + def test_cancelled_preparation_cannot_publish_late_snapshot(self): + version = self.store.cancel_preparing(self.claim.run_id) + with self.assertRaises(ConflictError): + self.store.complete_preparation(self.claim.run_id, self.claim.fencing_token, {"base": "late"}) + self.assertEqual(self.store.run(self.claim.run_id)["state"], "cancelled") + self.assertEqual(self.store.cancel_preparing(self.claim.run_id), version) + + def test_artifact_reference_is_versioned_and_terminal_run_is_immutable(self): + before = self.store.run(self.claim.run_id)["version"] + self.store.store_artifact(self.claim.run_id, "input.json", b"frozen input") + after = self.store.run(self.claim.run_id)["version"] + self.assertEqual(after, before + 1) + event = self.store.connection.execute( + "SELECT run_version FROM events WHERE run_id=? ORDER BY id DESC LIMIT 1", + (self.claim.run_id,), + ).fetchone() + self.assertEqual(event["run_version"], after) + terminal_version = self.store.cancel_preparing(self.claim.run_id) + with self.assertRaises(ConflictError): + self.store.store_artifact(self.claim.run_id, "late.json", b"late write") + self.assertEqual(self.store.run(self.claim.run_id)["version"], terminal_version) + count = self.store.connection.execute( + "SELECT COUNT(*) FROM artifacts WHERE run_id=?", (self.claim.run_id,), + ).fetchone()[0] + self.assertEqual(count, 1) + + +class StoreInitializationReview(unittest.TestCase): + def test_independent_processes_can_open_one_new_database(self): + context = multiprocessing.get_context("spawn") + with tempfile.TemporaryDirectory(prefix="devsquad-store-race-") as directory: + root = Path(directory) + barrier, results = context.Barrier(4), context.Queue() + processes = [context.Process(target=concurrent_open, args=(str(root / "db"), str(root / "artifacts"), barrier, results)) for _ in range(4)] + try: + for process in processes: + process.start() + outcomes = [results.get(timeout=15) for _ in processes] + for process in processes: + process.join(timeout=2) + self.assertEqual(outcomes, [None] * 4) + self.assertTrue(all(process.exitcode == 0 for process in processes)) + finally: + for process in processes: + if process.is_alive(): + process.terminate() + process.join(timeout=2) + results.close() + results.join_thread() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_store.py b/test/core/test_store.py new file mode 100644 index 0000000..5eee9d8 --- /dev/null +++ b/test/core/test_store.py @@ -0,0 +1,121 @@ +import json +from pathlib import Path +import sqlite3 +import subprocess +import tempfile +import threading +import unittest + +ROOT = Path(__file__).resolve().parents[2] +import sys +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.store import ConflictError, SchemaVersionError, Store, git_common_dir + + +class StoreTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True) + (self.repo / "README").write_text("base\n") + subprocess.run(["git", "-C", str(self.repo), "add", "README"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.database = self.root / "runtime/ledger.sqlite3" + self.artifacts = self.root / "runtime/artifacts" + self.store = Store(self.database, self.artifacts) + + def tearDown(self): + self.store.close() + self.temp.cleanup() + + def test_concurrent_identical_start_claims_one_run_and_conflicting_body_fails(self): + barrier = threading.Barrier(2) + results, errors = [], [] + def start(body): + connection = Store(self.database, self.artifacts) + try: + barrier.wait() + results.append(connection.claim_start(self.repo, "same-key", body, "owner")) + except Exception as exc: + errors.append(exc) + finally: + connection.close() + threads = [threading.Thread(target=start, args=({"task": "same", "n": 1},)) for _ in range(2)] + for thread in threads: thread.start() + for thread in threads: thread.join() + self.assertEqual(errors, []) + self.assertEqual(len({result.run_id for result in results}), 1) + self.assertEqual(sorted(result.created for result in results), [False, True]) + with self.assertRaises(ConflictError): + self.store.claim_start(self.repo, "same-key", {"task": "different"}, "owner-2") + + def test_request_is_claimed_before_snapshot_and_cancel_fences_late_preflight(self): + first = self.store.claim_start(self.repo, "k", {"task": "fixed"}, "owner") + run = self.store.run(first.run_id) + self.assertEqual((run["state"], run["phase"]), ("queued", "preparing")) + self.store.cancel_preparing(first.run_id) + with self.assertRaises(ConflictError): + self.store.complete_preparation(first.run_id, first.fencing_token, {"branch": "moved"}) + run = self.store.run(first.run_id) + self.assertEqual((run["state"], run["version"]), ("cancelled", 2)) + + def test_event_and_projection_compare_and_swap_share_transaction(self): + claim = self.store.claim_start(self.repo, "events", {"task": "x"}, "owner") + with self.assertRaises(ConflictError): + self.store.append_event(claim.run_id, 1, "unfenced", {}) + version = self.store.complete_preparation(claim.run_id, claim.fencing_token, {"head": "abc"}) + barrier = threading.Barrier(2) + successes, conflicts = [], [] + def mutate(label): + connection = Store(self.database, self.artifacts) + try: + barrier.wait() + successes.append(connection.append_event(claim.run_id, version, f"run.{label}", {"label": label})) + except ConflictError as exc: + conflicts.append(exc) + finally: + connection.close() + threads = [threading.Thread(target=mutate, args=(label,)) for label in ("a", "b")] + for thread in threads: thread.start() + for thread in threads: thread.join() + self.assertEqual(successes, [3]) + self.assertEqual(len(conflicts), 1) + events = self.store.connection.execute("SELECT run_version FROM events WHERE run_id=? ORDER BY id", (claim.run_id,)).fetchall() + self.assertEqual([row[0] for row in events], [1, 2, 3]) + + def test_git_common_dir_unifies_linked_worktrees(self): + linked = self.root / "linked" + subprocess.run(["git", "-C", str(self.repo), "worktree", "add", "-q", "-b", "linked", str(linked)], check=True) + self.assertEqual(git_common_dir(self.repo), git_common_dir(linked)) + a = self.store.claim_start(self.repo, "root", {"task": 1}, "a") + b = self.store.claim_start(linked, "linked", {"task": 2}, "b") + self.assertEqual(a.project_id, b.project_id) + + def test_artifact_is_finalized_and_verified_before_reference(self): + claim = self.store.claim_start(self.repo, "artifact", {"task": 1}, "owner") + path, digest, size = self.store.finalize_artifact(claim.run_id, "result.json", b'{"ok":true}') + self.assertEqual(size, path.stat().st_size) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM artifacts").fetchone()[0], 0) + path.write_bytes(b"tampered") + with self.assertRaises(ConflictError): + self.store.reference_artifact(claim.run_id, "result.json", path, digest) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM artifacts").fetchone()[0], 0) + artifact_id = self.store.store_artifact(claim.run_id, "final.json", b'{"done":true}') + row = self.store.connection.execute("SELECT sha256,byte_size FROM artifacts WHERE id=?", (artifact_id,)).fetchone() + self.assertEqual(row[1], 13) + + def test_migration_records_version_and_refuses_newer_database(self): + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 1) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(2,'future')") + self.store.close() + with self.assertRaises(SchemaVersionError): + Store(self.database, self.artifacts) + self.store = sqlite3.connect(":memory:") # tearDown-compatible close + + +if __name__ == "__main__": + unittest.main() From 606bc7b98313fb23c3a49c52aad358ac3eab7371 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 13:23:49 +0530 Subject: [PATCH 024/197] docs: checkpoint M2 store foundation --- docs/plans/engineering-team/M2-STATUS.md | 24 ++++++++++++++++++++++++ docs/plans/engineering-team/RESUME.md | 10 ++++++---- docs/plans/engineering-team/backlog.json | 14 ++++++++++++-- 3 files changed, 42 insertions(+), 6 deletions(-) create mode 100644 docs/plans/engineering-team/M2-STATUS.md diff --git a/docs/plans/engineering-team/M2-STATUS.md b/docs/plans/engineering-team/M2-STATUS.md new file mode 100644 index 0000000..d72372e --- /dev/null +++ b/docs/plans/engineering-team/M2-STATUS.md @@ -0,0 +1,24 @@ +# M2 implementation status + +M2 is **in progress**. The first store checkpoint is `0c2929c`; process +supervision, recovery and public start/status/cancel/resume service behavior +have not started. + +| Store requirement | Evidence | Status | +|---|---|---| +| SQLite WAL and packaged schema migration | Fresh install opens the packaged migration; four independent processes concurrently initialize one database | verified offline | +| Canonical request idempotency before mutable snapshot resolution | Concurrent connections return one run for the same key/body; a changed body conflicts | verified offline | +| Canonical project identity | Main and linked Git worktrees resolve to the same absolute common directory and project ID | verified offline | +| Preparing-owner fencing and cancellation | Runs remain `queued` with a private `preparing` phase; stale completion after cancel conflicts; generic events cannot bypass the fence | verified offline | +| Transactional projections and events | Compare-and-swap run version and append-only event commit together under concurrent writers | verified offline | +| Atomic hash-verified artifacts | Content-addressed files finalize before reference; references increment run version with an event; duplicates and terminal mutation cannot clobber prior content | verified offline | +| Unsupported future schema refusal | A database newer than migration version 1 is rejected | verified offline | + +Verification at this checkpoint: 46 core tests, including six independent M2 +review regressions; 10 legacy shell files with 202 assertions; and a temporary +wheel installation that applied the packaged migration. + +The next slice is the M2 supervisor and recovery foundation: durable service +operations, one supervisor claim and one active writer, process identity, +bounded output, cancellation/reaping, and crash reconciliation. This checkpoint +does not claim those behaviors or a working engineering workflow. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 990fc2d..c65f686 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -16,6 +16,8 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne `/Users/Dikshant/.devsquad/private-probes/native-codex-20260907T025419Z-97a10f0ae1e8`. - M1 is accepted for its bounded invocation/preparation scope. The next milestone is M2. +- M2 store foundation checkpoint: `0c2929c`. M2 remains in progress; its + supervisor, service operations and recovery slices are not implemented. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 has not started. @@ -50,10 +52,10 @@ The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed e ## Exact next work 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. -2. Read the M2 section of [IMPLEMENTATION.md](IMPLEMENTATION.md) and the - corresponding contracts before editing. Implement the bounded M2 store - foundation assigned by root, then the supervisor/recovery slices in their - dependency order. +2. Read the M2 section of [IMPLEMENTATION.md](IMPLEMENTATION.md), + [M2-STATUS.md](M2-STATUS.md), and the corresponding contracts before + editing. Continue with the bounded supervisor/recovery assignment from + root; do not redo the accepted store foundation. 3. Preserve M1 limitations and process-ownership boundaries. M1 acceptance is not a claim that the full product exists, and no additional native probe is needed for the accepted gate. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 68364ab..0ee7421 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -36,9 +36,19 @@ "id": "M2", "title": "Durable runs, cancellation and recovery", "depends_on": ["M1"], - "status": "pending", + "status": "in_progress", "acceptance_section": "M2 — Persist jobs and own their processes", - "evidence": [], + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "0c2929c", + "command_or_action": "46 core tests, 202 shell assertions, concurrent SQLite process tests and temporary installed-wheel migration", + "outcome": "Store foundation passes; supervisor, service API and recovery remain pending", + "artifact": "M2-STATUS.md", + "recorded_at": "2026-09-07T13:30:00+05:30", + "availability": "tracked tests" + } + ], "blocker": null }, { From ffc78e73024d6e9082143bc74185a4f5ecaf0ade Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 13:29:04 +0530 Subject: [PATCH 025/197] WIP checkpoint: start M2 supervisor lifecycle Shell suite passes. Core has one expected stale migration-version assertion while migration 002 and supervisor tests are unfinished. --- .../devsquad/migrations/002_supervisor.sql | 35 ++++ plugin/core/src/devsquad/store.py | 177 +++++++++++++++++- plugin/core/src/devsquad/supervisor.py | 163 ++++++++++++++++ 3 files changed, 369 insertions(+), 6 deletions(-) create mode 100644 plugin/core/src/devsquad/migrations/002_supervisor.sql create mode 100644 plugin/core/src/devsquad/supervisor.py diff --git a/plugin/core/src/devsquad/migrations/002_supervisor.sql b/plugin/core/src/devsquad/migrations/002_supervisor.sql new file mode 100644 index 0000000..13efc9f --- /dev/null +++ b/plugin/core/src/devsquad/migrations/002_supervisor.sql @@ -0,0 +1,35 @@ +ALTER TABLE runs ADD COLUMN worktree_path TEXT; + +CREATE TABLE supervisor_claims ( + run_id TEXT PRIMARY KEY REFERENCES runs(id), + owner_id TEXT NOT NULL, + fencing_token INTEGER NOT NULL, + package_digest TEXT NOT NULL, + heartbeat_at TEXT NOT NULL, + active INTEGER NOT NULL CHECK(active IN (0, 1)) +); + +CREATE TABLE attempts ( + id TEXT PRIMARY KEY, + run_id TEXT NOT NULL REFERENCES runs(id), + project_id TEXT NOT NULL REFERENCES projects(id), + worktree_path TEXT NOT NULL, + attempt_token TEXT NOT NULL UNIQUE, + status TEXT NOT NULL, + pid INTEGER, + pgid INTEGER, + process_start_id TEXT, + heartbeat_at TEXT NOT NULL, + package_digest TEXT NOT NULL, + stdout_artifact_id TEXT REFERENCES artifacts(id), + stderr_artifact_id TEXT REFERENCES artifacts(id), + output_metadata TEXT, + created_at TEXT NOT NULL, + finished_at TEXT +); + +CREATE UNIQUE INDEX one_active_writer_per_worktree +ON attempts(worktree_path) +WHERE status IN ('reserved', 'running', 'cancelling'); + +CREATE INDEX attempts_run ON attempts(run_id, created_at); diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index a56dcbd..e14c0b1 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import ContractError -SUPPORTED_SCHEMA_VERSION = 1 +SUPPORTED_SCHEMA_VERSION = 2 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} @@ -39,6 +39,15 @@ class StartClaim: fencing_token: int | None +@dataclass(frozen=True) +class AttemptReservation: + run_id: str + attempt_id: str + attempt_token: str + supervisor_token: int + version: int + + def _utc_now() -> str: return datetime.now(timezone.utc).isoformat() @@ -89,12 +98,14 @@ def migrate(self) -> None: current = self.connection.execute("SELECT COALESCE(MAX(version), 0) FROM schema_migrations").fetchone()[0] if table else 0 if current > SUPPORTED_SCHEMA_VERSION: raise SchemaVersionError(f"database schema {current} is newer than supported {SUPPORTED_SCHEMA_VERSION}") - if current == 0: - sql = files("devsquad.migrations").joinpath("001_initial.sql").read_text() + while current < SUPPORTED_SCHEMA_VERSION: + next_version = current + 1 + sql = files("devsquad.migrations").joinpath(f"{next_version:03d}_" + ("initial.sql" if next_version == 1 else "supervisor.sql")).read_text() for statement in sql.split(";"): if statement.strip(): self.connection.execute(statement) - self.connection.execute("INSERT INTO schema_migrations(version, applied_at) VALUES(1, ?)", (_utc_now(),)) + self.connection.execute("INSERT INTO schema_migrations(version, applied_at) VALUES(?, ?)", (next_version, _utc_now())) + current = next_version self.connection.execute("COMMIT") except Exception: self.connection.execute("ROLLBACK") @@ -128,8 +139,8 @@ def claim_start(self, worktree: Path, idempotency_key: str, submitted_request: A return StartClaim(existing["id"], project_id, digest, existing["version"], False, None) run_id, now = str(uuid.uuid4()), _utc_now() self.connection.execute( - "INSERT INTO runs(id,project_id,idempotency_key,request_hash,submitted_request,state,phase,version,created_at,updated_at) VALUES(?,?,?,?,?,'queued','preparing',1,?,?)", - (run_id, project_id, idempotency_key, digest, encoded, now, now), + "INSERT INTO runs(id,project_id,idempotency_key,request_hash,submitted_request,state,phase,version,created_at,updated_at,worktree_path) VALUES(?,?,?,?,?,'queued','preparing',1,?,?,?)", + (run_id, project_id, idempotency_key, digest, encoded, now, now, str(worktree.resolve(strict=True))), ) self.connection.execute( "INSERT INTO claims(run_id,kind,fencing_token,owner_id,active,claimed_at) VALUES(?,'preparing',1,?,1,?)", @@ -270,6 +281,160 @@ def store_artifact(self, run_id: str, name: str, content: bytes) -> str: path, digest, _ = self.finalize_artifact(run_id, name, content) return self.reference_artifact(run_id, name, path, digest) + def reserve_attempt(self, run_id: str, expected_version: int, owner_id: str, package_digest: str) -> AttemptReservation: + if not owner_id or not package_digest: + raise ContractError("supervisor owner and package digest are required") + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT project_id,worktree_path,state,phase,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not run or run["version"] != expected_version or run["state"] != "queued" or run["phase"] is not None: + raise ConflictError("run is not available for supervisor claim") + active = self.connection.execute("SELECT 1 FROM supervisor_claims WHERE run_id=? AND active=1", (run_id,)).fetchone() + if active: + raise ConflictError("run already has a supervisor claim") + if not run["worktree_path"]: + raise ContractError("run has no canonical worktree identity") + old = self.connection.execute("SELECT COALESCE(MAX(fencing_token),0) FROM supervisor_claims WHERE run_id=?", (run_id,)).fetchone()[0] + supervisor_token, attempt_id, attempt_token = old + 1, str(uuid.uuid4()), uuid.uuid4().hex + now, version = _utc_now(), expected_version + 1 + self.connection.execute("INSERT OR REPLACE INTO supervisor_claims(run_id,owner_id,fencing_token,package_digest,heartbeat_at,active) VALUES(?,?,?,?,?,1)", (run_id, owner_id, supervisor_token, package_digest, now)) + self.connection.execute("INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token,status,heartbeat_at,package_digest,created_at) VALUES(?,?,?,?,?,'reserved',?,?,?)", (attempt_id, run_id, run["project_id"], run["worktree_path"], attempt_token, now, package_digest, now)) + self.connection.execute("UPDATE runs SET phase='launching',version=?,updated_at=? WHERE id=?", (version, now, run_id)) + event = canonical_json({"attempt_id": attempt_id, "supervisor_token": supervisor_token}) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'supervisor.claimed',?,?)", (run_id, version, event, now)) + self.connection.execute("COMMIT") + return AttemptReservation(run_id, attempt_id, attempt_token, supervisor_token, version) + except sqlite3.IntegrityError as exc: + self.connection.execute("ROLLBACK") + raise ConflictError("worktree already has an active writer") from exc + except Exception: + self.connection.execute("ROLLBACK") + raise + + def mark_attempt_running(self, reservation: AttemptReservation, pid: int, pgid: int, process_start_id: str) -> int: + if any(type(value) is not int or value <= 0 for value in (pid, pgid)) or not process_start_id: + raise ContractError("valid process identity is required") + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT r.version,r.phase,a.status,a.attempt_token,s.fencing_token,s.active FROM runs r JOIN attempts a ON a.run_id=r.id JOIN supervisor_claims s ON s.run_id=r.id WHERE r.id=? AND a.id=?", (reservation.run_id, reservation.attempt_id)).fetchone() + if not row or row["version"] != reservation.version or row["phase"] != "launching" or row["status"] != "reserved" or row["attempt_token"] != reservation.attempt_token or row["fencing_token"] != reservation.supervisor_token or not row["active"]: + raise ConflictError("attempt reservation is stale") + version, now = row["version"] + 1, _utc_now() + self.connection.execute("UPDATE attempts SET status='running',pid=?,pgid=?,process_start_id=?,heartbeat_at=? WHERE id=?", (pid, pgid, process_start_id, now, reservation.attempt_id)) + self.connection.execute("UPDATE runs SET state='running',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, reservation.run_id)) + event = canonical_json({"attempt_id": reservation.attempt_id, "pid": pid, "pgid": pgid, "process_start_id": process_start_id}) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.running',?,?)", (reservation.run_id, version, event, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def heartbeat_attempt(self, run_id: str, attempt_token: str, supervisor_token: int) -> None: + now = _utc_now() + self.connection.execute("BEGIN IMMEDIATE") + try: + attempt = self.connection.execute("UPDATE attempts SET heartbeat_at=? WHERE run_id=? AND attempt_token=? AND status IN ('running','cancelling')", (now, run_id, attempt_token)).rowcount + claim = self.connection.execute("UPDATE supervisor_claims SET heartbeat_at=? WHERE run_id=? AND fencing_token=? AND active=1", (now, run_id, supervisor_token)).rowcount + if attempt != 1 or claim != 1: + raise ConflictError("attempt heartbeat is fenced") + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + + def request_cancel(self, run_id: str) -> tuple[int, dict[str, Any] | None]: + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not run: + raise ContractError("run does not exist") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"], None + if run["state"] == "cancelling": + attempt = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status='cancelling' ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + self.connection.execute("COMMIT") + return run["version"], dict(attempt) if attempt else None + if run["state"] != "running": + raise ConflictError("run is not cancellable by the supervisor") + attempt = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status='running'", (run_id,)).fetchone() + if not attempt: + raise ConflictError("running run has no active attempt") + version, now = run["version"] + 1, _utc_now() + self.connection.execute("UPDATE runs SET state='cancelling',version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("UPDATE attempts SET status='cancelling',heartbeat_at=? WHERE id=?", (now, attempt["id"])) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelling','{}',?)", (run_id, version, now)) + self.connection.execute("COMMIT") + return version, dict(attempt) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def finish_attempt(self, run_id: str, attempt_token: str, terminal_state: str, payload: Any) -> int: + if terminal_state not in TERMINAL_STATES: + raise ContractError("invalid terminal state") + encoded = canonical_json(payload) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + attempt = self.connection.execute("SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", (run_id, attempt_token)).fetchone() + if not run or not attempt or attempt["status"] not in {"running", "cancelling"}: + raise ConflictError("attempt completion is fenced") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"] + version, now = run["version"] + 1, _utc_now() + self.connection.execute("UPDATE attempts SET status='finished',finished_at=? WHERE id=?", (now, attempt["id"])) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("UPDATE runs SET state=?,phase=NULL,version=?,updated_at=? WHERE id=?", (terminal_state, version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", (run_id, version, f"run.{terminal_state}", encoded, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def record_attempt_output(self, run_id: str, attempt_token: str, stdout_id: str, stderr_id: str, metadata: Any) -> int: + encoded = canonical_json(metadata) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + attempt = self.connection.execute("SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", (run_id, attempt_token)).fetchone() + if not run or not attempt or attempt["status"] not in {"running", "cancelling"} or run["state"] in TERMINAL_STATES: + raise ConflictError("attempt output is fenced") + version, now = run["version"] + 1, _utc_now() + self.connection.execute("UPDATE attempts SET stdout_artifact_id=?,stderr_artifact_id=?,output_metadata=? WHERE id=?", (stdout_id, stderr_id, encoded, attempt["id"])) + self.connection.execute("UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'attempt.output',?,?)", (run_id, version, encoded, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def block_recovery(self, run_id: str, attempt_token: str, reason: str) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + attempt = self.connection.execute("SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", (run_id, attempt_token)).fetchone() + if not run or not attempt or attempt["status"] not in {"running", "cancelling"}: + raise ConflictError("recovery disposition is fenced") + version, now = run["version"] + 1, _utc_now() + self.connection.execute("UPDATE attempts SET status='recovery_required',finished_at=? WHERE id=?", (now, attempt["id"])) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("UPDATE runs SET state='blocked',phase='recovery_required',version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.blocked',?,?)", (run_id, version, canonical_json({"reason": reason, "next_action": "RECOVERY_REQUIRED"}), now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def active_attempt(self, run_id: str) -> dict[str, Any] | None: + row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status IN ('reserved','running','cancelling') ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + return dict(row) if row else None + def run(self, run_id: str) -> dict[str, Any]: row = self.connection.execute("SELECT * FROM runs WHERE id=?", (run_id,)).fetchone() if not row: raise ContractError("run does not exist") diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py new file mode 100644 index 0000000..503b04f --- /dev/null +++ b/plugin/core/src/devsquad/supervisor.py @@ -0,0 +1,163 @@ +"""Bounded process-group supervision for persisted M2 attempts.""" + +from __future__ import annotations + +from dataclasses import dataclass +import hashlib +import os +import signal +import subprocess +import threading +import time +from typing import Any, BinaryIO + +from .contracts import ContractError, LaunchSpec +from .store import AttemptReservation, ConflictError, Store + + +def process_start_identity(pid: int) -> str | None: + result = subprocess.run(["ps", "-p", str(pid), "-o", "lstart="], text=True, capture_output=True, check=False) + value = result.stdout.strip() + return value or None + + +def inspect_process(pid: int, pgid: int, expected_start: str) -> str: + observed = process_start_identity(pid) + if observed is None: + try: + os.killpg(pgid, 0) + except ProcessLookupError: + return "dead" + except PermissionError: + return "ambiguous" + return "ambiguous" + try: + observed_pgid = os.getpgid(pid) + except (ProcessLookupError, PermissionError): + return "ambiguous" + return "live" if observed == expected_start and observed_pgid == pgid else "ambiguous" + + +class BoundedDrain: + def __init__(self, stream: BinaryIO, limit: int): + self.stream, self.limit = stream, limit + self.content = bytearray() + self.total_bytes = 0 + self.digest = hashlib.sha256() + self.thread = threading.Thread(target=self._run, daemon=True) + + def _run(self) -> None: + while True: + chunk = self.stream.read(65536) + if not chunk: + return + self.total_bytes += len(chunk) + self.digest.update(chunk) + remaining = self.limit - len(self.content) + if remaining > 0: + self.content.extend(chunk[:remaining]) + + def start(self) -> None: + self.thread.start() + + def finish(self) -> dict[str, Any]: + self.thread.join(timeout=5) + if self.thread.is_alive(): + raise RuntimeError("output drain did not finish") + return {"total_bytes": self.total_bytes, "captured_bytes": len(self.content), "truncated": self.total_bytes > len(self.content), "full_sha256": self.digest.hexdigest()} + + +@dataclass +class RunningAttempt: + reservation: AttemptReservation + process: subprocess.Popen[bytes] + start_identity: str + stdout: BoundedDrain + stderr: BoundedDrain + + +class Supervisor: + def __init__(self, store: Store, *, output_limit: int = 1024 * 1024, grace_seconds: float = 5.0): + if output_limit <= 0 or grace_seconds < 0: + raise ContractError("supervisor bounds must be positive") + self.store, self.output_limit, self.grace_seconds = store, output_limit, grace_seconds + + def launch(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: str, package_digest: str) -> RunningAttempt: + reservation = self.store.reserve_attempt(run_id, expected_version, owner_id, package_digest) + environment = os.environ.copy(); environment.update(spec.environment) + process = subprocess.Popen(list(spec.argv), cwd=spec.cwd, env=environment, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True) + assert process.stdout is not None and process.stderr is not None + started = process_start_identity(process.pid) + if started is None: + process.wait(timeout=2) + raise RuntimeError("child exited before its identity could be persisted") + pgid = os.getpgid(process.pid) + self.store.mark_attempt_running(reservation, process.pid, pgid, started) + stdout, stderr = BoundedDrain(process.stdout, self.output_limit), BoundedDrain(process.stderr, self.output_limit) + stdout.start(); stderr.start() + return RunningAttempt(reservation, process, started, stdout, stderr) + + def _persist_output(self, handle: RunningAttempt) -> dict[str, Any]: + stdout_meta, stderr_meta = handle.stdout.finish(), handle.stderr.finish() + run_id, token = handle.reservation.run_id, handle.reservation.attempt_token + stdout_id = self.store.store_artifact(run_id, f"{handle.reservation.attempt_id}.stdout", bytes(handle.stdout.content)) + stderr_id = self.store.store_artifact(run_id, f"{handle.reservation.attempt_id}.stderr", bytes(handle.stderr.content)) + metadata = {"stdout": stdout_meta, "stderr": stderr_meta} + self.store.record_attempt_output(run_id, token, stdout_id, stderr_id, metadata) + return metadata + + def wait(self, handle: RunningAttempt, timeout_seconds: float) -> int: + try: + returncode = handle.process.wait(timeout=timeout_seconds) + except subprocess.TimeoutExpired: + self._terminate(handle) + metadata = self._persist_output(handle) + self.store.finish_attempt(handle.reservation.run_id, handle.reservation.attempt_token, "failed", {"error": "TIMEOUT", "output": metadata}) + return 124 + metadata = self._persist_output(handle) + terminal = "succeeded" if returncode == 0 else "failed" + self.store.finish_attempt(handle.reservation.run_id, handle.reservation.attempt_token, terminal, {"returncode": returncode, "output": metadata}) + return returncode + + def _terminate(self, handle: RunningAttempt) -> None: + attempt = self.store.active_attempt(handle.reservation.run_id) + if not attempt or inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) != "live": + raise ConflictError("process identity is not safe to signal") + os.killpg(attempt["pgid"], signal.SIGTERM) + try: + handle.process.wait(timeout=self.grace_seconds) + except subprocess.TimeoutExpired: + os.killpg(attempt["pgid"], signal.SIGKILL) + handle.process.wait(timeout=2) + deadline = time.monotonic() + 2 + while time.monotonic() < deadline: + try: os.killpg(attempt["pgid"], 0) + except ProcessLookupError: return + time.sleep(0.05) + raise RuntimeError("process group cleanup was not confirmed") + + def cancel(self, run_id: str, handle: RunningAttempt | None = None) -> int: + version, attempt = self.store.request_cancel(run_id) + if attempt is None: + return version + classification = inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) + if classification != "live": + if classification == "dead": + return self.store.block_recovery(run_id, attempt["attempt_token"], "child exited before cancellation cleanup") + return self.store.block_recovery(run_id, attempt["attempt_token"], "process identity is ambiguous or reused") + if handle is None or handle.reservation.attempt_token != attempt["attempt_token"]: + return version + self._terminate(handle) + metadata = self._persist_output(handle) + return self.store.finish_attempt(run_id, attempt["attempt_token"], "cancelled", {"output": metadata}) + + def recover(self, run_id: str) -> str: + attempt = self.store.active_attempt(run_id) + if not attempt: + raise ConflictError("run has no active attempt") + classification = inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) + if classification == "live": + return "live_owned" + reason = "confirmed dead child" if classification == "dead" else "process identity is ambiguous or reused" + self.store.block_recovery(run_id, attempt["attempt_token"], reason) + return "RECOVERY_REQUIRED" From 20bea665c3fa9cdfef91df8ddcfb5448448a1137 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Mon, 7 Sep 2026 13:29:23 +0530 Subject: [PATCH 026/197] docs: preserve interrupted M2 supervisor state --- docs/plans/engineering-team/RESUME.md | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index c65f686..0bd0aff 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -17,10 +17,15 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne - M1 is accepted for its bounded invocation/preparation scope. The next milestone is M2. - M2 store foundation checkpoint: `0c2929c`. M2 remains in progress; its - supervisor, service operations and recovery slices are not implemented. + supervisor, service operations and recovery slices are not accepted yet. +- Interrupted supervisor draft checkpoint: `ffc78e7`. It adds migration 002, + store-side attempt/claim operations and an initial supervisor, but has no + supervisor acceptance tests yet. The required shell suite passed before the + checkpoint. Core discovery had one expected failure: the store migration + test still asserted schema version 1 after the draft introduced version 2. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. -- Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 has not started. +- Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 is in progress. ## Completed and preserved @@ -54,8 +59,12 @@ The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed e 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. 2. Read the M2 section of [IMPLEMENTATION.md](IMPLEMENTATION.md), [M2-STATUS.md](M2-STATUS.md), and the corresponding contracts before - editing. Continue with the bounded supervisor/recovery assignment from - root; do not redo the accepted store foundation. + editing. Continue from `ffc78e7`: first repair the schema-version test and + add adversarial supervisor lifecycle tests. Review the draft for launch + failure rollback, crash/live-child recovery, identity ambiguity, bounded + output, TERM/KILL descendant cleanup, repeated cancellation and the + database-enforced single-writer rule. Do not redo the accepted store + foundation. 3. Preserve M1 limitations and process-ownership boundaries. M1 acceptance is not a claim that the full product exists, and no additional native probe is needed for the accepted gate. From 0ee4cc051dfafa620d3281c415dbaac1d23fbd6b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 8 Sep 2026 22:46:49 +0530 Subject: [PATCH 027/197] feat: complete fenced M2 supervisor lifecycle --- .../devsquad/migrations/002_supervisor.sql | 2 +- plugin/core/src/devsquad/store.py | 43 ++++- plugin/core/src/devsquad/supervisor.py | 132 ++++++++++++--- test/core/test_m2_supervisor_gate.py | 157 ++++++++++++++++++ test/core/test_store.py | 16 +- test/core/test_supervisor.py | 130 +++++++++++++++ 6 files changed, 446 insertions(+), 34 deletions(-) create mode 100644 test/core/test_m2_supervisor_gate.py create mode 100644 test/core/test_supervisor.py diff --git a/plugin/core/src/devsquad/migrations/002_supervisor.sql b/plugin/core/src/devsquad/migrations/002_supervisor.sql index 13efc9f..b9fe86f 100644 --- a/plugin/core/src/devsquad/migrations/002_supervisor.sql +++ b/plugin/core/src/devsquad/migrations/002_supervisor.sql @@ -30,6 +30,6 @@ CREATE TABLE attempts ( CREATE UNIQUE INDEX one_active_writer_per_worktree ON attempts(worktree_path) -WHERE status IN ('reserved', 'running', 'cancelling'); +WHERE status IN ('reserved', 'running', 'cancelling', 'ownership_ambiguous'); CREATE INDEX attempts_run ON attempts(run_id, created_at); diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index e14c0b1..035fd3b 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -82,11 +82,15 @@ def __init__(self, database: Path, artifacts: Path): artifacts.mkdir(parents=True, exist_ok=True) self.connection = sqlite3.connect(database, timeout=10, isolation_level=None) self.connection.row_factory = sqlite3.Row - self.connection.execute("PRAGMA busy_timeout=10000") - self.connection.execute("PRAGMA foreign_keys=ON") - self.migrate() - self.connection.execute("PRAGMA journal_mode=WAL") - self.connection.execute("PRAGMA synchronous=FULL") + try: + self.connection.execute("PRAGMA busy_timeout=10000") + self.connection.execute("PRAGMA foreign_keys=ON") + self.migrate() + self.connection.execute("PRAGMA journal_mode=WAL") + self.connection.execute("PRAGMA synchronous=FULL") + except Exception: + self.connection.close() + raise def close(self) -> None: self.connection.close() @@ -228,7 +232,8 @@ def finalize_artifact(self, run_id: str, name: str, content: bytes) -> tuple[Pat digest = hashlib.sha256(content).hexdigest() directory = self.artifacts / run_id directory.mkdir(parents=True, exist_ok=True) - destination = directory / f"{digest}.blob" + name_key = hashlib.sha256(name.encode()).hexdigest()[:16] + destination = directory / f"{digest}.{name_key}.blob" descriptor, temporary = tempfile.mkstemp(prefix=f".{name}.", dir=directory) try: with os.fdopen(descriptor, "wb") as stream: @@ -413,7 +418,7 @@ def record_attempt_output(self, run_id: str, attempt_token: str, stdout_id: str, self.connection.execute("ROLLBACK") raise - def block_recovery(self, run_id: str, attempt_token: str, reason: str) -> int: + def block_recovery(self, run_id: str, attempt_token: str, reason: str, *, release_writer: bool = False) -> int: self.connection.execute("BEGIN IMMEDIATE") try: run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() @@ -421,7 +426,8 @@ def block_recovery(self, run_id: str, attempt_token: str, reason: str) -> int: if not run or not attempt or attempt["status"] not in {"running", "cancelling"}: raise ConflictError("recovery disposition is fenced") version, now = run["version"] + 1, _utc_now() - self.connection.execute("UPDATE attempts SET status='recovery_required',finished_at=? WHERE id=?", (now, attempt["id"])) + status = "recovery_required" if release_writer else "ownership_ambiguous" + self.connection.execute("UPDATE attempts SET status=?,finished_at=? WHERE id=?", (status, now, attempt["id"])) self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) self.connection.execute("UPDATE runs SET state='blocked',phase='recovery_required',version=?,updated_at=? WHERE id=?", (version, now, run_id)) self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.blocked',?,?)", (run_id, version, canonical_json({"reason": reason, "next_action": "RECOVERY_REQUIRED"}), now)) @@ -431,8 +437,27 @@ def block_recovery(self, run_id: str, attempt_token: str, reason: str) -> int: self.connection.execute("ROLLBACK") raise + def fail_launch(self, reservation: AttemptReservation, reason: str) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT phase,version FROM runs WHERE id=?", (reservation.run_id,)).fetchone() + attempt = self.connection.execute("SELECT status,attempt_token FROM attempts WHERE id=?", (reservation.attempt_id,)).fetchone() + if not run or run["phase"] != "launching" or not attempt or attempt["status"] != "reserved" or attempt["attempt_token"] != reservation.attempt_token: + raise ConflictError("launch failure disposition is fenced") + version, now = run["version"] + 1, _utc_now() + self.connection.execute("UPDATE attempts SET status='recovery_required',finished_at=? WHERE id=?", (now, reservation.attempt_id)) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=? AND fencing_token=?", (reservation.run_id, reservation.supervisor_token)) + self.connection.execute("UPDATE runs SET state='blocked',phase='recovery_required',version=?,updated_at=? WHERE id=?", (version, now, reservation.run_id)) + payload = canonical_json({"reason": reason, "next_action": "RECOVERY_REQUIRED"}) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.blocked',?,?)", (reservation.run_id, version, payload, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + def active_attempt(self, run_id: str) -> dict[str, Any] | None: - row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status IN ('reserved','running','cancelling') ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status IN ('reserved','running','cancelling','ownership_ambiguous') ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() return dict(row) if row else None def run(self, run_id: str) -> dict[str, Any]: diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 503b04f..ab5ab8d 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -3,10 +3,12 @@ from __future__ import annotations from dataclasses import dataclass +import ctypes import hashlib import os import signal import subprocess +import sys import threading import time from typing import Any, BinaryIO @@ -16,9 +18,25 @@ def process_start_identity(pid: int) -> str | None: - result = subprocess.run(["ps", "-p", str(pid), "-o", "lstart="], text=True, capture_output=True, check=False) - value = result.stdout.strip() - return value or None + if sys.platform.startswith("linux"): + try: + fields = open(f"/proc/{pid}/stat", encoding="ascii").read().rsplit(") ", 1)[1].split() + return f"linux-start-ticks:{fields[19]}" + except (OSError, IndexError): + return None + if sys.platform == "darwin": + class ProcBsdInfo(ctypes.Structure): + _fields_ = [("prefix", ctypes.c_byte * 120), ("start_sec", ctypes.c_uint64), ("start_usec", ctypes.c_uint64)] + info = ProcBsdInfo() + try: + function = ctypes.CDLL("/usr/lib/libproc.dylib").proc_pidinfo + function.argtypes = [ctypes.c_int, ctypes.c_int, ctypes.c_uint64, ctypes.c_void_p, ctypes.c_int] + function.restype = ctypes.c_int + copied = function(pid, 3, 0, ctypes.byref(info), ctypes.sizeof(info)) + except OSError: + return None + return f"darwin-start:{info.start_sec}:{info.start_usec}" if copied == ctypes.sizeof(info) else None + return None def inspect_process(pid: int, pgid: int, expected_start: str) -> str: @@ -38,6 +56,15 @@ def inspect_process(pid: int, pgid: int, expected_start: str) -> str: return "live" if observed == expected_start and observed_pgid == pgid else "ambiguous" +def _live_group_exists(pgid: int) -> bool: + result = subprocess.run(["ps", "-axo", "pgid=,stat="], text=True, capture_output=True, check=False) + for line in result.stdout.splitlines(): + fields = line.split() + if len(fields) >= 2 and fields[0].isdigit() and int(fields[0]) == pgid and not fields[1].startswith("Z"): + return True + return False + + class BoundedDrain: def __init__(self, stream: BinaryIO, limit: int): self.stream, self.limit = stream, limit @@ -64,6 +91,7 @@ def finish(self) -> dict[str, Any]: self.thread.join(timeout=5) if self.thread.is_alive(): raise RuntimeError("output drain did not finish") + self.stream.close() return {"total_bytes": self.total_bytes, "captured_bytes": len(self.content), "truncated": self.total_bytes > len(self.content), "full_sha256": self.digest.hexdigest()} @@ -85,14 +113,46 @@ def __init__(self, store: Store, *, output_limit: int = 1024 * 1024, grace_secon def launch(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: str, package_digest: str) -> RunningAttempt: reservation = self.store.reserve_attempt(run_id, expected_version, owner_id, package_digest) environment = os.environ.copy(); environment.update(spec.environment) - process = subprocess.Popen(list(spec.argv), cwd=spec.cwd, env=environment, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True) + try: + stdin_stream: Any = subprocess.DEVNULL + if spec.stdin_path is not None: + stdin_path = os.path.realpath(spec.stdin_path) + if os.path.islink(spec.stdin_path) or not os.path.isfile(stdin_path): + raise ContractError("stdin artifact must be a regular non-symlink file") + stdin_stream = open(stdin_path, "rb") + try: + process = subprocess.Popen(list(spec.argv), cwd=spec.cwd, env=environment, stdin=stdin_stream, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True) + finally: + if stdin_stream is not subprocess.DEVNULL: + stdin_stream.close() + except Exception: + self.store.fail_launch(reservation, "process spawn failed") + raise assert process.stdout is not None and process.stderr is not None started = process_start_identity(process.pid) if started is None: - process.wait(timeout=2) - raise RuntimeError("child exited before its identity could be persisted") - pgid = os.getpgid(process.pid) - self.store.mark_attempt_running(reservation, process.pid, pgid, started) + returncode = process.poll() + if returncode is None: + try: os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: pass + process.wait(timeout=2) + process.stdout.close(); process.stderr.close() + self.store.fail_launch(reservation, "live child had no strong process identity") + raise RuntimeError("live child identity could not be persisted") + started = f"exited-before-observation:{returncode}:non-signalable" + pgid = process.pid + else: + pgid = os.getpgid(process.pid) + try: + self.store.mark_attempt_running(reservation, process.pid, pgid, started) + except Exception: + if process.poll() is None: + try: os.killpg(pgid, signal.SIGKILL) + except ProcessLookupError: pass + process.wait(timeout=2) + process.stdout.close(); process.stderr.close() + self.store.fail_launch(reservation, "process identity could not be persisted") + raise stdout, stderr = BoundedDrain(process.stdout, self.output_limit), BoundedDrain(process.stderr, self.output_limit) stdout.start(); stderr.start() return RunningAttempt(reservation, process, started, stdout, stderr) @@ -107,34 +167,62 @@ def _persist_output(self, handle: RunningAttempt) -> dict[str, Any]: return metadata def wait(self, handle: RunningAttempt, timeout_seconds: float) -> int: - try: - returncode = handle.process.wait(timeout=timeout_seconds) - except subprocess.TimeoutExpired: + deadline = time.monotonic() + timeout_seconds + returncode = None + while returncode is None and time.monotonic() < deadline: + returncode = handle.process.poll() + if returncode is None: + self.store.heartbeat_attempt(handle.reservation.run_id, handle.reservation.attempt_token, handle.reservation.supervisor_token) + time.sleep(min(0.2, max(0, deadline - time.monotonic()))) + if returncode is None: self._terminate(handle) metadata = self._persist_output(handle) self.store.finish_attempt(handle.reservation.run_id, handle.reservation.attempt_token, "failed", {"error": "TIMEOUT", "output": metadata}) return 124 + self._cleanup_owned_group(handle) metadata = self._persist_output(handle) terminal = "succeeded" if returncode == 0 else "failed" self.store.finish_attempt(handle.reservation.run_id, handle.reservation.attempt_token, terminal, {"returncode": returncode, "output": metadata}) return returncode + def _cleanup_owned_group(self, handle: RunningAttempt) -> None: + pgid = handle.process.pid + handle.process.poll() + if not _live_group_exists(pgid): return + os.killpg(pgid, signal.SIGTERM) + deadline = time.monotonic() + self.grace_seconds + while time.monotonic() < deadline: + handle.process.poll() + if not _live_group_exists(pgid): return + time.sleep(0.05) + try: os.killpg(pgid, signal.SIGKILL) + except ProcessLookupError: return + deadline = time.monotonic() + 2 + while time.monotonic() < deadline: + handle.process.poll() + if not _live_group_exists(pgid): return + time.sleep(0.05) + raise RuntimeError("owned process group remains after KILL") + def _terminate(self, handle: RunningAttempt) -> None: attempt = self.store.active_attempt(handle.reservation.run_id) if not attempt or inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) != "live": raise ConflictError("process identity is not safe to signal") os.killpg(attempt["pgid"], signal.SIGTERM) - try: - handle.process.wait(timeout=self.grace_seconds) - except subprocess.TimeoutExpired: - os.killpg(attempt["pgid"], signal.SIGKILL) - handle.process.wait(timeout=2) + deadline = time.monotonic() + self.grace_seconds + while time.monotonic() < deadline: + handle.process.poll() + if not _live_group_exists(attempt["pgid"]): return + time.sleep(0.05) + try: os.killpg(attempt["pgid"], signal.SIGKILL) + except ProcessLookupError: return + try: handle.process.wait(timeout=2) + except subprocess.TimeoutExpired: pass deadline = time.monotonic() + 2 while time.monotonic() < deadline: - try: os.killpg(attempt["pgid"], 0) - except ProcessLookupError: return + if not _live_group_exists(attempt["pgid"]): return time.sleep(0.05) - raise RuntimeError("process group cleanup was not confirmed") + raise RuntimeError("process group cleanup was not confirmed after KILL") def cancel(self, run_id: str, handle: RunningAttempt | None = None) -> int: version, attempt = self.store.request_cancel(run_id) @@ -143,10 +231,10 @@ def cancel(self, run_id: str, handle: RunningAttempt | None = None) -> int: classification = inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) if classification != "live": if classification == "dead": - return self.store.block_recovery(run_id, attempt["attempt_token"], "child exited before cancellation cleanup") + return self.store.block_recovery(run_id, attempt["attempt_token"], "child exited before cancellation cleanup", release_writer=True) return self.store.block_recovery(run_id, attempt["attempt_token"], "process identity is ambiguous or reused") if handle is None or handle.reservation.attempt_token != attempt["attempt_token"]: - return version + return self.store.block_recovery(run_id, attempt["attempt_token"], "live child requires owner reconciliation before cancellation") self._terminate(handle) metadata = self._persist_output(handle) return self.store.finish_attempt(run_id, attempt["attempt_token"], "cancelled", {"output": metadata}) @@ -159,5 +247,5 @@ def recover(self, run_id: str) -> str: if classification == "live": return "live_owned" reason = "confirmed dead child" if classification == "dead" else "process identity is ambiguous or reused" - self.store.block_recovery(run_id, attempt["attempt_token"], reason) + self.store.block_recovery(run_id, attempt["attempt_token"], reason, release_writer=classification == "dead") return "RECOVERY_REQUIRED" diff --git a/test/core/test_m2_supervisor_gate.py b/test/core/test_m2_supervisor_gate.py new file mode 100644 index 0000000..1a4c49e --- /dev/null +++ b/test/core/test_m2_supervisor_gate.py @@ -0,0 +1,157 @@ +"""Independent adversarial gates for M2 process supervision.""" +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import signal +import shutil +import subprocess +import sys +import tempfile +import time +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) + +from devsquad.contracts import ExecutionIdentity, LaunchSpec +from devsquad.store import ConflictError, Store +from devsquad.supervisor import Supervisor, inspect_process, process_start_identity + + +class SupervisorGateReview(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-supervisor-gate-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + self.store = Store(self.root / "runtime.sqlite3", self.root / "artifacts") + self.addCleanup(self.store.close) + self.supervisor = Supervisor(self.store, output_limit=32, grace_seconds=0.1) + + def _ready_run(self, key: str): + claim = self.store.claim_start(self.repo, key, {"task": key}, "preflight") + version = self.store.complete_preparation(claim.run_id, claim.fencing_token, {"head": "fixed"}) + return claim.run_id, version + + def _spec(self, *argv: str, stdin_path: str | None = None) -> LaunchSpec: + return LaunchSpec( + 1, "fake", "cli_exec", tuple(argv), str(self.repo), stdin_path, 5, + ExecutionIdentity("fake", "1", "fixture", "fixture", "fixture", "low"), + ) + + @staticmethod + def _force_cleanup(handle) -> None: + if handle.process.poll() is None: + try: + os.killpg(handle.process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + handle.process.wait(timeout=2) + for drain in (handle.stdout, handle.stderr): + drain.thread.join(timeout=2) + drain.stream.close() + + def test_current_process_has_stable_strong_identity(self): + first = process_start_identity(os.getpid()) + second = process_start_identity(os.getpid()) + self.assertIsInstance(first, str) + self.assertEqual(first, second) + self.assertEqual(inspect_process(os.getpid(), os.getpgid(os.getpid()), first), "live") + + def test_fast_success_is_a_valid_attempt(self): + run_id, version = self._ready_run("fast") + true_binary = shutil.which("true") + self.assertIsNotNone(true_binary) + handle = self.supervisor.launch(run_id, version, self._spec(true_binary), "owner", "package") + self.assertEqual(self.supervisor.wait(handle, 2), 0) + self.assertEqual(self.store.run(run_id)["state"], "succeeded") + + def test_spawn_failure_is_fenced_and_releases_worktree_writer(self): + run_id, version = self._ready_run("missing") + with self.assertRaises(FileNotFoundError): + self.supervisor.launch(run_id, version, self._spec("/definitely/missing/devsquad-worker"), "owner", "package") + self.assertEqual(self.store.run(run_id)["state"], "blocked") + self.assertIsNone(self.store.active_attempt(run_id)) + second, second_version = self._ready_run("after-missing") + reservation = self.store.reserve_attempt(second, second_version, "other-owner", "package") + self.assertTrue(reservation.attempt_token) + + def test_output_is_bounded_but_full_stream_is_accounted(self): + run_id, version = self._ready_run("output") + code = "import sys;sys.stdout.write('o'*1000);sys.stderr.write('e'*2000)" + handle = self.supervisor.launch(run_id, version, self._spec(sys.executable, "-c", code), "owner", "package") + self.assertEqual(self.supervisor.wait(handle, 3), 0) + attempt = self.store.connection.execute("SELECT * FROM attempts WHERE run_id=?", (run_id,)).fetchone() + metadata = json.loads(attempt["output_metadata"]) + self.assertEqual(metadata["stdout"], { + "total_bytes": 1000, "captured_bytes": 32, "truncated": True, + "full_sha256": hashlib.sha256(b"o" * 1000).hexdigest(), + }) + self.assertEqual(metadata["stderr"]["total_bytes"], 2000) + self.assertEqual(metadata["stderr"]["captured_bytes"], 32) + for column in ("stdout_artifact_id", "stderr_artifact_id"): + artifact = self.store.connection.execute( + "SELECT path,byte_size FROM artifacts WHERE id=?", (attempt[column],), + ).fetchone() + self.assertEqual(artifact["byte_size"], 32) + self.assertEqual(Path(artifact["path"]).stat().st_size, 32) + + def test_database_fences_second_writer_for_same_worktree(self): + first_run, first_version = self._ready_run("writer-one") + handle = self.supervisor.launch( + first_run, first_version, self._spec(sys.executable, "-c", "import time;time.sleep(30)"), + "owner-one", "package", + ) + self.addCleanup(self._force_cleanup, handle) + second_run, second_version = self._ready_run("writer-two") + with self.assertRaises(ConflictError): + self.store.reserve_attempt(second_run, second_version, "owner-two", "package") + self.supervisor.cancel(first_run, handle) + self.assertEqual(self.store.run(first_run)["state"], "cancelled") + + def test_recovery_never_signals_an_ambiguous_identity(self): + run_id, version = self._ready_run("ambiguous") + handle = self.supervisor.launch( + run_id, version, self._spec(sys.executable, "-c", "import time;time.sleep(30)"), + "owner", "package", + ) + self.addCleanup(self._force_cleanup, handle) + with mock.patch("devsquad.supervisor.process_start_identity", return_value="different-start"), \ + mock.patch("devsquad.supervisor.os.killpg") as killpg: + self.assertEqual(self.supervisor.recover(run_id), "RECOVERY_REQUIRED") + killpg.assert_not_called() + self.assertEqual(self.store.run(run_id)["state"], "blocked") + attempt = self.store.connection.execute( + "SELECT status FROM attempts WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(attempt["status"], "ownership_ambiguous") + second, second_version = self._ready_run("after-ambiguous") + with self.assertRaises(ConflictError): + self.store.reserve_attempt(second, second_version, "other-owner", "package") + + def test_confirmed_dead_recovery_releases_writer_fence(self): + run_id, version = self._ready_run("dead") + handle = self.supervisor.launch( + run_id, version, self._spec(sys.executable, "-c", "pass"), "owner", "package", + ) + self.addCleanup(self._force_cleanup, handle) + handle.process.wait(timeout=2) + handle.stdout.thread.join(timeout=2) + handle.stderr.thread.join(timeout=2) + self.assertEqual(self.supervisor.recover(run_id), "RECOVERY_REQUIRED") + attempt = self.store.connection.execute( + "SELECT status FROM attempts WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(attempt["status"], "recovery_required") + second, second_version = self._ready_run("after-dead") + reservation = self.store.reserve_attempt(second, second_version, "other-owner", "package") + self.assertTrue(reservation.attempt_token) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_store.py b/test/core/test_store.py index 5eee9d8..70cdb9e 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -109,13 +109,25 @@ def test_artifact_is_finalized_and_verified_before_reference(self): self.assertEqual(row[1], 13) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 1) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(2,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 2) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(3,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) self.store = sqlite3.connect(":memory:") # tearDown-compatible close + def test_version_one_fixture_migrates_to_version_two(self): + old_db = self.root / "old.sqlite3" + connection = sqlite3.connect(old_db) + sql = (ROOT / "plugin/core/src/devsquad/migrations/001_initial.sql").read_text() + connection.executescript(sql) + connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(1,'fixture')") + connection.commit(); connection.close() + upgraded = Store(old_db, self.root / "old-artifacts") + self.addCleanup(upgraded.close) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 2) + self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) + if __name__ == "__main__": unittest.main() diff --git a/test/core/test_supervisor.py b/test/core/test_supervisor.py new file mode 100644 index 0000000..cbf8807 --- /dev/null +++ b/test/core/test_supervisor.py @@ -0,0 +1,130 @@ +import json +import os +from pathlib import Path +import signal +import subprocess +import sys +import tempfile +import time +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ExecutionIdentity, LaunchSpec +from devsquad.store import ConflictError, Store +from devsquad.supervisor import Supervisor, inspect_process, process_start_identity + + +class SupervisorTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-supervisor-") + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + self.store = Store(self.root / "state.sqlite3", self.root / "artifacts") + self.supervisor = Supervisor(self.store, output_limit=32, grace_seconds=0.1) + self.handles = [] + + def tearDown(self): + for handle in self.handles: + if handle.process.poll() is None: + try: os.killpg(handle.process.pid, signal.SIGKILL) + except ProcessLookupError: pass + handle.process.wait(timeout=2) + handle.stdout.thread.join(timeout=1); handle.stderr.thread.join(timeout=1) + if not handle.stdout.stream.closed: handle.stdout.stream.close() + if not handle.stderr.stream.closed: handle.stderr.stream.close() + self.store.close(); self.temp.cleanup() + + def prepared(self, key, argv): + claim = self.store.claim_start(self.repo, key, {"task": key}, "prepare") + version = self.store.complete_preparation(claim.run_id, claim.fencing_token, {"head": "fixed"}) + identity = ExecutionIdentity("fake", "1", None, None, None, None) + spec = LaunchSpec(1, "fake", "cli_exec", tuple(argv), str(self.repo), None, 5, identity, {}) + return claim.run_id, version, spec + + def launch(self, key, code): + run_id, version, spec = self.prepared(key, (sys.executable, "-c", code)) + handle = self.supervisor.launch(run_id, version, spec, "supervisor", "package-sha") + self.handles.append(handle) + return run_id, handle + + def test_strong_process_identity_is_stable_and_live(self): + first = process_start_identity(os.getpid()) + self.assertIsNotNone(first) + self.assertEqual(first, process_start_identity(os.getpid())) + self.assertEqual(inspect_process(os.getpid(), os.getpgrp(), first), "live") + + def test_normal_and_fast_completion_persist_bounded_output(self): + run_id, handle = self.launch("output", "import sys; print('o'*100); print('e'*100,file=sys.stderr)") + self.assertEqual(self.supervisor.wait(handle, 3), 0) + run = self.store.run(run_id) + self.assertEqual(run["state"], "succeeded") + attempt = self.store.connection.execute("SELECT * FROM attempts WHERE run_id=?", (run_id,)).fetchone() + metadata = json.loads(attempt["output_metadata"]) + self.assertTrue(metadata["stdout"]["truncated"] and metadata["stderr"]["truncated"]) + for column in ("stdout_artifact_id", "stderr_artifact_id"): + artifact = self.store.connection.execute("SELECT byte_size FROM artifacts WHERE id=?", (attempt[column],)).fetchone() + self.assertLessEqual(artifact[0], 32) + fast_id, version, spec = self.prepared("fast", ("/usr/bin/true",)) + fast = self.supervisor.launch(fast_id, version, spec, "supervisor", "package-sha") + self.handles.append(fast) + self.assertEqual(self.supervisor.wait(fast, 2), 0) + self.assertEqual(self.store.run(fast_id)["state"], "succeeded") + + def test_one_active_writer_per_worktree(self): + first_id, first = self.launch("one", "import time; time.sleep(30)") + second_id, version, spec = self.prepared("two", (sys.executable, "-c", "print('never')")) + with self.assertRaises(ConflictError): + self.supervisor.launch(second_id, version, spec, "other", "package-sha") + self.supervisor.cancel(first_id, first) + + def test_timeout_kills_term_ignoring_root_and_descendant(self): + code = """import os,signal,time +signal.signal(signal.SIGTERM, signal.SIG_IGN) +if os.fork()==0: + signal.signal(signal.SIGTERM, signal.SIG_IGN) + while True: time.sleep(1) +while True: time.sleep(1) +""" + run_id, handle = self.launch("timeout", code) + pgid = handle.process.pid + self.assertEqual(self.supervisor.wait(handle, 0.2), 124) + with self.assertRaises(ProcessLookupError): os.killpg(pgid, 0) + self.assertEqual(self.store.run(run_id)["state"], "failed") + + def test_cancel_is_idempotent_after_confirmed_cleanup(self): + run_id, handle = self.launch("cancel", "import time; time.sleep(30)") + version = self.supervisor.cancel(run_id, handle) + self.assertEqual(self.store.run(run_id)["state"], "cancelled") + self.assertEqual(self.supervisor.cancel(run_id), version) + + def test_recovery_live_dead_and_reused_identity_never_relaunches_or_signals(self): + live_id, live = self.launch("live", "import time; time.sleep(30)") + self.assertEqual(Supervisor(self.store).recover(live_id), "live_owned") + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM attempts WHERE run_id=?", (live_id,)).fetchone()[0], 1) + self.supervisor.cancel(live_id, live) + + dead_id, dead = self.launch("dead", "pass") + dead.process.wait(timeout=2); dead.stdout.thread.join(1); dead.stderr.thread.join(1) + self.assertEqual(self.supervisor.recover(dead_id), "RECOVERY_REQUIRED") + self.assertEqual(self.store.run(dead_id)["state"], "blocked") + + reused_id, reused = self.launch("reused", "import time; time.sleep(30)") + self.store.connection.execute("UPDATE attempts SET process_start_id='different' WHERE run_id=?", (reused_id,)) + with mock.patch("devsquad.supervisor.os.killpg") as signal_group: + self.assertEqual(self.supervisor.recover(reused_id), "RECOVERY_REQUIRED") + signal_group.assert_not_called() + os.killpg(reused.process.pid, signal.SIGKILL); reused.process.wait(timeout=2) + + def test_spawn_failure_leaves_safe_blocked_reservation(self): + run_id, version, spec = self.prepared("spawn-fail", (str(self.root / "missing"),)) + with self.assertRaises(OSError): + self.supervisor.launch(run_id, version, spec, "supervisor", "package-sha") + self.assertEqual((self.store.run(run_id)["state"], self.store.run(run_id)["phase"]), ("blocked", "recovery_required")) + self.assertIsNone(self.store.active_attempt(run_id)) + + +if __name__ == "__main__": unittest.main() From df955f4dc9dfcc30b7d343cf15dca3de1967ada0 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 8 Sep 2026 22:49:09 +0530 Subject: [PATCH 028/197] fix: fail closed on supervisor cleanup races --- plugin/core/src/devsquad/supervisor.py | 23 +++++++++++++++++++---- test/core/test_supervisor.py | 7 ++++++- 2 files changed, 25 insertions(+), 5 deletions(-) diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index ab5ab8d..b7bea59 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -57,11 +57,19 @@ def inspect_process(pid: int, pgid: int, expected_start: str) -> str: def _live_group_exists(pgid: int) -> bool: - result = subprocess.run(["ps", "-axo", "pgid=,stat="], text=True, capture_output=True, check=False) + result = subprocess.run(["/bin/ps", "-axo", "pgid=,stat="], text=True, capture_output=True, check=False) + if result.returncode != 0: + raise RuntimeError("process-group inventory failed") + parsed = 0 for line in result.stdout.splitlines(): fields = line.split() - if len(fields) >= 2 and fields[0].isdigit() and int(fields[0]) == pgid and not fields[1].startswith("Z"): + if len(fields) < 2 or not fields[0].isdigit(): + raise RuntimeError("process-group inventory was malformed") + parsed += 1 + if int(fields[0]) == pgid and not fields[1].startswith("Z"): return True + if parsed == 0: + raise RuntimeError("process-group inventory was empty") return False @@ -175,7 +183,11 @@ def wait(self, handle: RunningAttempt, timeout_seconds: float) -> int: self.store.heartbeat_attempt(handle.reservation.run_id, handle.reservation.attempt_token, handle.reservation.supervisor_token) time.sleep(min(0.2, max(0, deadline - time.monotonic()))) if returncode is None: - self._terminate(handle) + try: + self._terminate(handle) + except ConflictError: + self.store.block_recovery(handle.reservation.run_id, handle.reservation.attempt_token, "timeout raced with process identity change") + raise metadata = self._persist_output(handle) self.store.finish_attempt(handle.reservation.run_id, handle.reservation.attempt_token, "failed", {"error": "TIMEOUT", "output": metadata}) return 124 @@ -235,7 +247,10 @@ def cancel(self, run_id: str, handle: RunningAttempt | None = None) -> int: return self.store.block_recovery(run_id, attempt["attempt_token"], "process identity is ambiguous or reused") if handle is None or handle.reservation.attempt_token != attempt["attempt_token"]: return self.store.block_recovery(run_id, attempt["attempt_token"], "live child requires owner reconciliation before cancellation") - self._terminate(handle) + try: + self._terminate(handle) + except ConflictError: + return self.store.block_recovery(run_id, attempt["attempt_token"], "cancellation raced with process identity change") metadata = self._persist_output(handle) return self.store.finish_attempt(run_id, attempt["attempt_token"], "cancelled", {"output": metadata}) diff --git a/test/core/test_supervisor.py b/test/core/test_supervisor.py index cbf8807..ee3f0a3 100644 --- a/test/core/test_supervisor.py +++ b/test/core/test_supervisor.py @@ -14,7 +14,7 @@ from devsquad.contracts import ExecutionIdentity, LaunchSpec from devsquad.store import ConflictError, Store -from devsquad.supervisor import Supervisor, inspect_process, process_start_identity +from devsquad.supervisor import Supervisor, _live_group_exists, inspect_process, process_start_identity class SupervisorTest(unittest.TestCase): @@ -57,6 +57,11 @@ def test_strong_process_identity_is_stable_and_live(self): self.assertEqual(first, process_start_identity(os.getpid())) self.assertEqual(inspect_process(os.getpid(), os.getpgrp(), first), "live") + def test_process_inventory_failure_never_means_absence(self): + failed = subprocess.CompletedProcess(["/bin/ps"], 1, "", "denied") + with mock.patch("devsquad.supervisor.subprocess.run", return_value=failed), self.assertRaises(RuntimeError): + _live_group_exists(123) + def test_normal_and_fast_completion_persist_bounded_output(self): run_id, handle = self.launch("output", "import sys; print('o'*100); print('e'*100,file=sys.stderr)") self.assertEqual(self.supervisor.wait(handle, 3), 0) From 31452f4e482e79d702c07b17321378871d40fcac Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 8 Sep 2026 22:49:58 +0530 Subject: [PATCH 029/197] docs: checkpoint M2 supervisor lifecycle --- docs/plans/engineering-team/M2-STATUS.md | 26 ++++++++++++++---------- docs/plans/engineering-team/RESUME.md | 23 ++++++++++----------- docs/plans/engineering-team/backlog.json | 8 ++++---- 3 files changed, 30 insertions(+), 27 deletions(-) diff --git a/docs/plans/engineering-team/M2-STATUS.md b/docs/plans/engineering-team/M2-STATUS.md index d72372e..202e7d7 100644 --- a/docs/plans/engineering-team/M2-STATUS.md +++ b/docs/plans/engineering-team/M2-STATUS.md @@ -1,8 +1,8 @@ # M2 implementation status -M2 is **in progress**. The first store checkpoint is `0c2929c`; process -supervision, recovery and public start/status/cancel/resume service behavior -have not started. +M2 is **in progress**. The store checkpoint is `0c2929c`; the bounded +supervisor lifecycle checkpoints are `0ee4cc0` and `df955f4`. Public detached service +operations and end-to-end shell persistence have not started. | Store requirement | Evidence | Status | |---|---|---| @@ -12,13 +12,17 @@ have not started. | Preparing-owner fencing and cancellation | Runs remain `queued` with a private `preparing` phase; stale completion after cancel conflicts; generic events cannot bypass the fence | verified offline | | Transactional projections and events | Compare-and-swap run version and append-only event commit together under concurrent writers | verified offline | | Atomic hash-verified artifacts | Content-addressed files finalize before reference; references increment run version with an event; duplicates and terminal mutation cannot clobber prior content | verified offline | -| Unsupported future schema refusal | A database newer than migration version 1 is rejected | verified offline | +| Schema migration and future refusal | A real version-1 fixture upgrades to version 2; newer unsupported versions are rejected | verified offline | +| Supervisor and writer fencing | Transactional claims allow one supervisor and one active writer per worktree; ambiguous ownership retains the database fence | verified offline | +| Strong process identity and recovery | Darwin start second+microsecond identity is stable; live children remain owned without relaunch; dead and reused identities receive distinct recovery dispositions and reused IDs are never signalled | verified offline | +| Bounded process lifecycle | Direct argv runs in a new session; heartbeat, PID/PGID/start identity, token and package digest persist; stdout/stderr drain continuously with truncation and full-stream hashes | verified offline | +| Timeout and cancellation | Intent precedes verified TERM/KILL; TERM-resistant root and descendant disappear before completion; repeated terminal cancel is harmless | verified offline | -Verification at this checkpoint: 46 core tests, including six independent M2 -review regressions; 10 legacy shell files with 202 assertions; and a temporary -wheel installation that applied the packaged migration. +Verification at this checkpoint: 62 core tests, including fourteen supervisor +tests and seven independent supervisor regressions; 10 legacy shell files with +202 assertions; and a temporary wheel installation that applied migration 2 +and imported the supervisor. -The next slice is the M2 supervisor and recovery foundation: durable service -operations, one supervisor claim and one active writer, process identity, -bounded output, cancellation/reaping, and crash reconciliation. This checkpoint -does not claim those behaviors or a working engineering workflow. +The next slice is durable start/status/events/result/cancel/resume service +behavior and a supervisor process that survives the launching shell. This +checkpoint does not claim those behaviors or a working engineering workflow. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 0bd0aff..6a55d19 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,7 +2,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 7, 2026 +## Current position — September 8, 2026 - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` is the published runtime baseline. @@ -18,11 +18,13 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne milestone is M2. - M2 store foundation checkpoint: `0c2929c`. M2 remains in progress; its supervisor, service operations and recovery slices are not accepted yet. -- Interrupted supervisor draft checkpoint: `ffc78e7`. It adds migration 002, - store-side attempt/claim operations and an initial supervisor, but has no - supervisor acceptance tests yet. The required shell suite passed before the - checkpoint. Core discovery had one expected failure: the store migration - test still asserted schema version 1 after the draft introduced version 2. +- Supervisor lifecycle checkpoint: `0ee4cc0`. It completes migration 002, + fenced supervisor/writer ownership, strong process identity, bounded output, + heartbeat, timeout/cancel cleanup and conservative recovery. M2 remains in + progress because durable service operations and a detached supervisor entry + point are pending. +- Cleanup-race hardening checkpoint: `df955f4`. Cleanup inventory fails closed, + timeout/cancel identity races retain ownership fencing, and 62 core tests pass. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 is in progress. @@ -59,12 +61,9 @@ The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed e 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. 2. Read the M2 section of [IMPLEMENTATION.md](IMPLEMENTATION.md), [M2-STATUS.md](M2-STATUS.md), and the corresponding contracts before - editing. Continue from `ffc78e7`: first repair the schema-version test and - add adversarial supervisor lifecycle tests. Review the draft for launch - failure rollback, crash/live-child recovery, identity ambiguity, bounded - output, TERM/KILL descendant cleanup, repeated cancellation and the - database-enforced single-writer rule. Do not redo the accepted store - foundation. + editing. Continue from `df955f4` with only the bounded durable service and + detached-supervisor assignment supplied by root. Do not redo the accepted + store or supervisor lifecycle foundations. 3. Preserve M1 limitations and process-ownership boundaries. M1 acceptance is not a claim that the full product exists, and no additional native probe is needed for the accepted gate. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 0ee7421..b1c36ba 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -41,11 +41,11 @@ "evidence": [ { "kind": "implementation_checkpoint", - "revision": "0c2929c", - "command_or_action": "46 core tests, 202 shell assertions, concurrent SQLite process tests and temporary installed-wheel migration", - "outcome": "Store foundation passes; supervisor, service API and recovery remain pending", + "revision": "df955f4", + "command_or_action": "62 core tests, 202 shell assertions, real subprocess lifecycle tests, migration-1 fixture upgrade and installed-wheel migration 2", + "outcome": "Store and bounded supervisor lifecycle pass; durable detached service operations remain pending", "artifact": "M2-STATUS.md", - "recorded_at": "2026-09-07T13:30:00+05:30", + "recorded_at": "2026-09-08T00:00:00+05:30", "availability": "tracked tests" } ], From cefd1936a7a83987e94235094034e2cab681e880 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 8 Sep 2026 22:57:58 +0530 Subject: [PATCH 030/197] WIP checkpoint: add M2 durable service surface --- docs/plans/engineering-team/RESUME.md | 7 ++ plugin/core/src/devsquad/cli.py | 30 +++++- plugin/core/src/devsquad/detached.py | 33 +++++++ plugin/core/src/devsquad/fake_step.py | 10 ++ plugin/core/src/devsquad/service.py | 135 ++++++++++++++++++++++++++ plugin/core/src/devsquad/store.py | 29 ++++++ test/core/test_service.py | 100 +++++++++++++++++++ 7 files changed, 343 insertions(+), 1 deletion(-) create mode 100644 plugin/core/src/devsquad/detached.py create mode 100644 plugin/core/src/devsquad/fake_step.py create mode 100644 plugin/core/src/devsquad/service.py create mode 100644 test/core/test_service.py diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 6a55d19..e69a3ef 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -25,6 +25,13 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne point are pending. - Cleanup-race hardening checkpoint: `df955f4`. Cleanup inventory fails closed, timeout/cancel identity races retain ownership fencing, and 62 core tests pass. +- Service/CLI draft is currently uncommitted WIP. It adds strict public + start/status/events/result/cancel/resume shapes and a detached entrypoint, + but must be redesigned before acceptance: worker output needs durable spool + files, launch needs an exec gate tied to persisted identity, detached wait + must observe cancel intent, and migration discovery must be ordered rather + than filename-hardcoded. Preflight failure, predecessor validation, event + cursors and recovery-file semantics also remain open. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 is in progress. diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 5ca4ab4..7fcdd0a 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -4,12 +4,15 @@ import argparse import json +import os import sys from pathlib import Path from . import __version__ from .adapters import AdapterManifest, classify_cli, harness_version, prepare_cli, prepare_native_codex_from_catalog from .contracts import ContractError, envelope, error_payload +from .service import Service +from .store import ConflictError SOURCE_ROOT = Path(__file__).resolve().parents[2] CORE_ROOT = SOURCE_ROOT if (SOURCE_ROOT / "adapters").is_dir() else Path(sys.prefix) / "share" / "devsquad" @@ -54,6 +57,24 @@ def command_classify(args: argparse.Namespace) -> dict: return envelope(data=result.to_dict()), 0 +def _service(args: argparse.Namespace) -> Service: + return Service(Path(args.runtime_dir)) + + +def command_start(args: argparse.Namespace) -> tuple[dict, int]: + task = json.loads(Path(args.task_file).read_text()) + return envelope(data=_service(args).start(task, args.idempotency_key, args.supersedes_run)), 0 + + +def command_status(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).status(args.run)), 0 +def command_events(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).events(args.run, args.after, args.limit)), 0 +def command_result(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).result(args.run)), 0 +def command_cancel(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).cancel(args.run)), 0 +def command_resume(args: argparse.Namespace) -> tuple[dict, int]: + recovery = json.loads(Path(args.recovery_file).read_text()) if args.recovery_file else None + return envelope(data=_service(args).resume(args.run, recovery)), 0 + + class ContractParser(argparse.ArgumentParser): def error(self, message: str) -> None: raise ContractError(message) @@ -81,6 +102,13 @@ def parser() -> argparse.ArgumentParser: cmd.add_argument("--stdout-file", required=True) cmd.add_argument("--stderr-file", required=True) cmd.set_defaults(func=fn) + runtime_default = os.environ.get("DEVSQUAD_RUNTIME_DIR", str(Path.home() / ".devsquad" / "runtime")) + start = sub.add_parser("start"); start.add_argument("--task-file", required=True); start.add_argument("--idempotency-key", required=True); start.add_argument("--supersedes-run"); start.add_argument("--json", action="store_true"); start.add_argument("--runtime-dir", default=runtime_default); start.set_defaults(func=command_start) + for name, fn in (("status",command_status),("result",command_result),("cancel",command_cancel),("resume",command_resume)): + cmd=sub.add_parser(name); cmd.add_argument("run"); cmd.add_argument("--json",action="store_true"); cmd.add_argument("--runtime-dir",default=runtime_default) + if name == "resume": cmd.add_argument("--recovery-file") + cmd.set_defaults(func=fn) + events=sub.add_parser("events"); events.add_argument("run"); events.add_argument("--after",type=int,default=0); events.add_argument("--limit",type=int,default=100); events.add_argument("--json",action="store_true"); events.add_argument("--runtime-dir",default=runtime_default); events.set_defaults(func=command_events) return p @@ -93,7 +121,7 @@ def main(argv: list[str] | None = None) -> int: except (ContractError, OSError, json.JSONDecodeError) as exc: code = getattr(exc, "code", "INPUT_INVALID") print(json.dumps(envelope(error=error_payload(code, str(exc))), sort_keys=True)) - return 64 + return 75 if isinstance(exc, ConflictError) else 64 if __name__ == "__main__": diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py new file mode 100644 index 0000000..338dbc5 --- /dev/null +++ b/plugin/core/src/devsquad/detached.py @@ -0,0 +1,33 @@ +"""Detached M2 supervisor entrypoint; invoked only from the durable service.""" +import argparse +import json +import os +from pathlib import Path +import sys + +from .contracts import ExecutionIdentity, LaunchSpec +from .store import ConflictError, Store +from .supervisor import Supervisor + + +def main(argv=None): + parser = argparse.ArgumentParser() + parser.add_argument("--database", required=True); parser.add_argument("--artifacts", required=True) + parser.add_argument("--run-id", required=True); parser.add_argument("--expected-version", type=int, required=True) + parser.add_argument("--package-digest", required=True) + args = parser.parse_args(argv) + store = Store(Path(args.database), Path(args.artifacts)) + try: + run = store.run(args.run_id); snapshot = json.loads(run["mutable_snapshot"]) + environment = {"DEVSQUAD_WORKER": "1", "DEVSQUAD_RUN_ID": args.run_id} + identity = ExecutionIdentity("devsquad-fake-step", "1", None, None, None, None) + command = [sys.executable, "-m", "devsquad.fake_step"] + if "internal_fake_delay" in snapshot: command += ["--delay", str(snapshot["internal_fake_delay"])] + spec = LaunchSpec(1, "devsquad-fake-step", "cli_exec", tuple(command), run["worktree_path"], None, snapshot["task"]["budget"]["wall_seconds"], identity, environment) + supervisor = Supervisor(store) + try: handle = supervisor.launch(args.run_id, args.expected_version, spec, f"daemon:{os.getpid()}", args.package_digest) + except ConflictError: return 0 + return 0 if supervisor.wait(handle, spec.timeout_seconds) == 0 else 1 + finally: store.close() + +if __name__ == "__main__": raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/fake_step.py b/plugin/core/src/devsquad/fake_step.py new file mode 100644 index 0000000..a2f5dab --- /dev/null +++ b/plugin/core/src/devsquad/fake_step.py @@ -0,0 +1,10 @@ +"""Internal deterministic lifecycle fixture. It is not a public task command.""" +import argparse +import sys +import time + +parser = argparse.ArgumentParser() +parser.add_argument("--delay", type=float, default=0.05) +delay = parser.parse_args().delay +time.sleep(delay) +sys.stdout.write("M2_FAKE_STEP_OK\n") diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py new file mode 100644 index 0000000..e11739c --- /dev/null +++ b/plugin/core/src/devsquad/service.py @@ -0,0 +1,135 @@ +"""Durable M2 application operations shared by CLI and later MCP surfaces.""" +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +from typing import Any + +from .contracts import ContractError +from .store import ConflictError, Store, TERMINAL_STATES, canonical_json +from .validation import validate_task + + +class Service: + def __init__(self, runtime: Path): + self.runtime = runtime.resolve() + self.runtime.mkdir(parents=True, exist_ok=True) + self.database = self.runtime / "state.sqlite3" + self.artifacts = self.runtime / "artifacts" + + def _store(self) -> Store: + return Store(self.database, self.artifacts) + + def _freeze_package(self) -> tuple[Path, str]: + source = Path(__file__).resolve().parent + digest = hashlib.sha256() + members = sorted(p for p in source.rglob("*") if p.is_file() and "__pycache__" not in p.parts and p.suffix in {".py", ".sql"}) + for path in members: + relative = path.relative_to(source) + digest.update(str(relative).encode() + b"\0" + path.read_bytes()) + value = digest.hexdigest() + destination = self.runtime / "packages" / value / "devsquad" + if not destination.exists(): + destination.parent.parent.mkdir(parents=True, exist_ok=True) + temporary = Path(tempfile.mkdtemp(prefix=f".{value}.", dir=destination.parent.parent)) + try: + shutil.copytree(source, temporary / "devsquad", ignore=shutil.ignore_patterns("__pycache__", "*.pyc")) + destination.parent.mkdir(parents=True, exist_ok=True) + os.replace(temporary / "devsquad", destination) + finally: + shutil.rmtree(temporary, ignore_errors=True) + return destination.parent, value + + @staticmethod + def _resolve_snapshot(task: dict[str, Any], internal_delay: float | None) -> dict[str, Any]: + repo = Path(task["project"]["repo_path"]).resolve(strict=True) + def oid(ref: str) -> str: + result = subprocess.run(["git", "-C", str(repo), "rev-parse", "--verify", f"{ref}^{{commit}}"], text=True, capture_output=True, check=False) + if result.returncode != 0: raise ContractError(f"Git ref does not resolve to a commit: {ref}") + return result.stdout.strip() + configs = {} + for label in ("profiles_file", "policy_file"): + candidate = Path(task["routing"][label]) + path = (repo / candidate).resolve() if not candidate.is_absolute() else candidate.resolve() + if path != repo and repo not in path.parents: raise ContractError(f"{label} escapes project") + data = path.read_bytes(); configs[label] = {"path": str(path), "sha256": hashlib.sha256(data).hexdigest()} + snapshot = {"task": task, "base_oid": oid(task["project"]["base_ref"]), "target_oid": oid(task["project"]["target_ref"]), "configs": configs} + if internal_delay is not None: + if internal_delay < 0 or internal_delay > 60: raise ContractError("internal fake delay is invalid") + snapshot["internal_fake_delay"] = internal_delay + return snapshot + + def start(self, task: dict[str, Any], idempotency_key: str, supersedes_run_id: str | None = None, *, _internal_fake_delay: float | None = None) -> dict[str, Any]: + validate_task(task, require_existing_repo=True) + submitted = {"task": task, "supersedes_run_id": supersedes_run_id} + store = self._store() + try: + claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}") + if not claim.created: + return {"run_id": claim.run_id, "state": store.run(claim.run_id)["state"], "created": False} + snapshot = self._resolve_snapshot(task, _internal_fake_delay) + version = store.complete_preparation(claim.run_id, claim.fencing_token or 0, snapshot) + finally: + store.close() + package, digest = self._freeze_package() + self._spawn_daemon(claim.run_id, version, package, digest) + return {"run_id": claim.run_id, "state": "queued", "created": True} + + def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, digest: str) -> int: + command = [sys.executable, "-m", "devsquad.detached", "--database", str(self.database), "--artifacts", str(self.artifacts), "--run-id", run_id, "--expected-version", str(expected_version), "--package-digest", digest] + environment = {"PATH": os.environ.get("PATH", ""), "PYTHONPATH": str(package)} + process = subprocess.Popen(command, cwd=self.runtime, env=environment, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, start_new_session=True, close_fds=True) + return process.pid + + def status(self, run_id: str) -> dict[str, Any]: + store = self._store() + try: + run = store.run(run_id); attempt = store.active_attempt(run_id) + return {"run_id": run_id, "state": run["state"], "phase": run["phase"], "version": run["version"], "active_attempt": {k: attempt.get(k) for k in ("id","status","pid","pgid","heartbeat_at")} if attempt else None, "next_action": "recovery_file_required" if run["state"] == "blocked" else None} + finally: store.close() + + def events(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any]: + store = self._store() + try: return store.events_page(run_id, after, limit) + finally: store.close() + + def result(self, run_id: str) -> dict[str, Any]: + store = self._store() + try: + run = store.run(run_id) + return {"run_id": run_id, "ready": run["state"] in TERMINAL_STATES, "state": run["state"], "artifacts": store.artifacts_for_run(run_id) if run["state"] in TERMINAL_STATES else []} + finally: store.close() + + def cancel(self, run_id: str) -> dict[str, Any]: + store = self._store() + try: + run = store.run(run_id) + if run["state"] == "queued" and run["phase"] is None: version = store.cancel_queued(run_id) + elif run["state"] in {"running", "cancelling"}: version, _ = store.request_cancel(run_id) + elif run["state"] in TERMINAL_STATES: version = run["version"] + else: raise ConflictError("run requires recovery before cancellation") + return {"run_id": run_id, "state": store.run(run_id)["state"], "version": version} + finally: store.close() + + def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[str, Any]: + store = self._store() + try: + run = store.run(run_id) + if run["state"] in TERMINAL_STATES: raise ConflictError("terminal run cannot resume; start a superseding run") + if run["state"] == "running": + from .supervisor import Supervisor + disposition = Supervisor(store).recover(run_id) + return {"run_id": run_id, "disposition": disposition, "launched": False} + if run["state"] == "queued" and run["phase"] is None: + version = run["version"] + else: + raise ConflictError("run requires an explicit recovery disposition") + finally: store.close() + package, digest = self._freeze_package(); self._spawn_daemon(run_id, version, package, digest) + return {"run_id": run_id, "disposition": "continued", "launched": True} diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 035fd3b..772188a 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -460,6 +460,35 @@ def active_attempt(self, run_id: str) -> dict[str, Any] | None: row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status IN ('reserved','running','cancelling','ownership_ambiguous') ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() return dict(row) if row else None + def cancel_queued(self, run_id: str) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute("SELECT state,phase,version FROM runs WHERE id=?", (run_id,)).fetchone() + if not row: + raise ContractError("run does not exist") + if row["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT"); return row["version"] + if row["state"] != "queued" or row["phase"] is not None: + raise ConflictError("queued run is owned by another operation") + version, now = row["version"] + 1, _utc_now() + self.connection.execute("UPDATE runs SET state='cancelled',version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelled','{}',?)", (run_id, version, now)) + self.connection.execute("COMMIT"); return version + except Exception: + self.connection.execute("ROLLBACK"); raise + + def events_page(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any]: + if type(after) is not int or after < 0 or type(limit) is not int or not 1 <= limit <= 1000: + raise ContractError("event cursor/limit is invalid") + if not self.connection.execute("SELECT 1 FROM runs WHERE id=?", (run_id,)).fetchone(): + raise ContractError("run does not exist") + rows = self.connection.execute("SELECT id,run_version,type,payload,created_at FROM events WHERE run_id=? AND id>? ORDER BY id LIMIT ?", (run_id, after, limit + 1)).fetchall() + page, more = rows[:limit], len(rows) > limit + return {"events": [{**dict(row), "payload": json.loads(row["payload"])} for row in page], "next_cursor": page[-1]["id"] if more and page else None} + + def artifacts_for_run(self, run_id: str) -> list[dict[str, Any]]: + return [dict(row) for row in self.connection.execute("SELECT id,name,path,sha256,byte_size,created_at FROM artifacts WHERE run_id=? ORDER BY created_at,id", (run_id,))] + def run(self, run_id: str) -> dict[str, Any]: row = self.connection.execute("SELECT * FROM runs WHERE id=?", (run_id,)).fetchone() if not row: raise ContractError("run does not exist") diff --git a/test/core/test_service.py b/test/core/test_service.py new file mode 100644 index 0000000..9e2ba65 --- /dev/null +++ b/test/core/test_service.py @@ -0,0 +1,100 @@ +import json +import os +from pathlib import Path +import signal +import subprocess +import sys +import tempfile +import time +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.service import Service +from devsquad.store import ConflictError, Store + + +class ServiceTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-service-") + self.root = Path(self.temp.name); self.repo = self.root / "repo"; self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True) + (self.repo / "profiles.json").write_text("{}\n"); (self.repo / "policy.json").write_text("{}\n") + subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + self.task["project"] = {"repo_path": str(self.repo), "base_ref": "HEAD", "target_ref": "HEAD"} + self.task["routing"]["profiles_file"] = "profiles.json"; self.task["routing"]["policy_file"] = "policy.json" + self.service = Service(self.runtime) + + def tearDown(self): self.temp.cleanup() + + def wait_state(self, run_id, states, timeout=8): + deadline=time.monotonic()+timeout + while time.monotonic() Date: Wed, 9 Sep 2026 09:53:20 +0530 Subject: [PATCH 031/197] WIP checkpoint: gate durable M2 attempt runner --- docs/plans/engineering-team/RESUME.md | 25 ++++--- plugin/core/src/devsquad/attempt_runner.py | 52 ++++++++++++++ plugin/core/src/devsquad/detached.py | 4 +- .../devsquad/migrations/003_durable_io.sql | 6 ++ plugin/core/src/devsquad/service.py | 11 ++- plugin/core/src/devsquad/store.py | 48 +++++++++++-- plugin/core/src/devsquad/supervisor.py | 72 +++++++++++++++++++ test/core/test_service.py | 6 +- test/core/test_store.py | 8 +-- 9 files changed, 206 insertions(+), 26 deletions(-) create mode 100644 plugin/core/src/devsquad/attempt_runner.py create mode 100644 plugin/core/src/devsquad/migrations/003_durable_io.sql diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index e69a3ef..8680031 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -25,13 +25,18 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne point are pending. - Cleanup-race hardening checkpoint: `df955f4`. Cleanup inventory fails closed, timeout/cancel identity races retain ownership fencing, and 62 core tests pass. -- Service/CLI draft is currently uncommitted WIP. It adds strict public - start/status/events/result/cancel/resume shapes and a detached entrypoint, - but must be redesigned before acceptance: worker output needs durable spool - files, launch needs an exec gate tied to persisted identity, detached wait - must observe cancel intent, and migration discovery must be ordered rather - than filename-hardcoded. Preflight failure, predecessor validation, event - cursors and recovery-file semantics also remain open. +- Durable service checkpoint `cefd193` is explicit WIP. The following local + checkpoint replaces the unsafe anonymous-pipe launch with a persisted, + gated attempt runner and migration 003 durable paths. The runner cannot + launch the internal fixture worker until its strong identity and spool paths + are committed; it owns output drains, heartbeat, cancel polling and an + atomic exit record. Ordered migration discovery, transactional status, + stable event cursors and fenced preparation failure are implemented. + Recovery/import of a runner exit record, predecessor validation, package + pin verification, result-receipt barrier and the full cross-process crash + matrix remain open. Public branch-review execution is explicitly failed as + `CAPABILITY_UNAVAILABLE` until M3; only the internal test hook can run the + fake step. M2 remains in progress. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 is in progress. @@ -68,9 +73,9 @@ The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed e 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. 2. Read the M2 section of [IMPLEMENTATION.md](IMPLEMENTATION.md), [M2-STATUS.md](M2-STATUS.md), and the corresponding contracts before - editing. Continue from `df955f4` with only the bounded durable service and - detached-supervisor assignment supplied by root. Do not redo the accepted - store or supervisor lifecycle foundations. + editing. Continue by implementing durable receipt import/recovery without + relaunch, then package/predecessor/result invariants and the cross-process + crash tests. Do not redo the accepted store or supervisor foundations. 3. Preserve M1 limitations and process-ownership boundaries. M1 acceptance is not a claim that the full product exists, and no additional native probe is needed for the accepted gate. diff --git a/plugin/core/src/devsquad/attempt_runner.py b/plugin/core/src/devsquad/attempt_runner.py new file mode 100644 index 0000000..ff48f35 --- /dev/null +++ b/plugin/core/src/devsquad/attempt_runner.py @@ -0,0 +1,52 @@ +"""Frozen, gated worker owner that survives its detached coordinator.""" +import argparse, hashlib, json, os, signal, subprocess, threading, time +from pathlib import Path +from .store import Store +from .supervisor import process_start_identity + +def _atomic(path: Path, value): + temporary=path.with_name(path.name+f".tmp.{os.getpid()}"); temporary.write_text(json.dumps(value,sort_keys=True)+"\n"); os.replace(temporary,path) + +def _drain(stream, capture: Path, limit: int, result: dict): + digest=hashlib.sha256(); total=0; kept=0 + with capture.open("wb") as output: + while True: + chunk=stream.read(65536) + if not chunk: break + total+=len(chunk); digest.update(chunk); remaining=limit-kept + if remaining>0: output.write(chunk[:remaining]); kept+=min(remaining,len(chunk)) + output.flush(); os.fsync(output.fileno()) + result.update(total_bytes=total,captured_bytes=kept,truncated=total>kept,full_sha256=digest.hexdigest()) + +def main(argv=None): + p=argparse.ArgumentParser(); p.add_argument("--gate-fd",type=int,required=True); p.add_argument("--database",type=Path,required=True); p.add_argument("--artifacts",type=Path,required=True); p.add_argument("--run-id",required=True); p.add_argument("--attempt-token",required=True); p.add_argument("--supervisor-token",type=int,required=True); p.add_argument("--stdout",type=Path,required=True); p.add_argument("--stderr",type=Path,required=True); p.add_argument("--exit-record",type=Path,required=True); p.add_argument("--child-record",type=Path,required=True); p.add_argument("--limit",type=int,required=True); p.add_argument("command",nargs=argparse.REMAINDER) + a=p.parse_args(argv); command=a.command[1:] if a.command[:1]==["--"] else a.command + with os.fdopen(a.gate_fd,"rb",closefd=True) as gate: + if gate.read(1)!=b"1": return 125 + child=subprocess.Popen(command,stdout=subprocess.PIPE,stderr=subprocess.PIPE,start_new_session=True) + started=process_start_identity(child.pid) + if started is None: + child.kill(); child.wait(); return 126 + _atomic(a.child_record,{"pid":child.pid,"pgid":child.pid,"process_start_id":started}) + out,err={},{}; threads=[threading.Thread(target=_drain,args=(child.stdout,a.stdout,a.limit,out)),threading.Thread(target=_drain,args=(child.stderr,a.stderr,a.limit,err))] + for thread in threads: thread.start() + store=Store(a.database,a.artifacts); cancelled=False + try: + while child.poll() is None: + run=store.run(a.run_id) + if run["state"]=="cancelling": + cancelled=True + try: os.killpg(child.pid,signal.SIGTERM) + except ProcessLookupError: pass + try: child.wait(timeout=5) + except subprocess.TimeoutExpired: + try: os.killpg(child.pid,signal.SIGKILL) + except ProcessLookupError: pass + break + store.heartbeat_attempt(a.run_id,a.attempt_token,a.supervisor_token); time.sleep(.1) + returncode=child.wait() + finally: store.close() + for thread in threads: thread.join() + _atomic(a.exit_record,{"returncode":returncode,"cancelled":cancelled,"stdout":out,"stderr":err,"finished_at":time.time()}) + return returncode +if __name__=="__main__": raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index 338dbc5..831d369 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -25,9 +25,9 @@ def main(argv=None): if "internal_fake_delay" in snapshot: command += ["--delay", str(snapshot["internal_fake_delay"])] spec = LaunchSpec(1, "devsquad-fake-step", "cli_exec", tuple(command), run["worktree_path"], None, snapshot["task"]["budget"]["wall_seconds"], identity, environment) supervisor = Supervisor(store) - try: handle = supervisor.launch(args.run_id, args.expected_version, spec, f"daemon:{os.getpid()}", args.package_digest) + try: handle = supervisor.launch_durable(args.run_id, args.expected_version, spec, f"daemon:{os.getpid()}", args.package_digest) except ConflictError: return 0 - return 0 if supervisor.wait(handle, spec.timeout_seconds) == 0 else 1 + return 0 if supervisor.wait_durable(handle, spec.timeout_seconds) == 0 else 1 finally: store.close() if __name__ == "__main__": raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/migrations/003_durable_io.sql b/plugin/core/src/devsquad/migrations/003_durable_io.sql new file mode 100644 index 0000000..0002d01 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/003_durable_io.sql @@ -0,0 +1,6 @@ +ALTER TABLE attempts ADD COLUMN stdout_spool TEXT; +ALTER TABLE attempts ADD COLUMN stderr_spool TEXT; +ALTER TABLE attempts ADD COLUMN stdout_meta TEXT; +ALTER TABLE attempts ADD COLUMN stderr_meta TEXT; +ALTER TABLE attempts ADD COLUMN exit_record TEXT; +ALTER TABLE attempts ADD COLUMN child_record TEXT; diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index e11739c..177712c 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -73,7 +73,14 @@ def start(self, task: dict[str, Any], idempotency_key: str, supersedes_run_id: s claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}") if not claim.created: return {"run_id": claim.run_id, "state": store.run(claim.run_id)["state"], "created": False} - snapshot = self._resolve_snapshot(task, _internal_fake_delay) + try: + snapshot = self._resolve_snapshot(task, _internal_fake_delay) + except Exception as exc: + store.fail_preparation(claim.run_id, claim.fencing_token or 0, {"error": "PREPARATION_FAILED", "message": str(exc)}) + raise + if _internal_fake_delay is None: + store.fail_preparation(claim.run_id, claim.fencing_token or 0, {"error": "CAPABILITY_UNAVAILABLE", "message": "branch-review workflow is introduced in M3"}) + return {"run_id": claim.run_id, "state": "failed", "created": True} version = store.complete_preparation(claim.run_id, claim.fencing_token or 0, snapshot) finally: store.close() @@ -90,7 +97,7 @@ def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, diges def status(self, run_id: str) -> dict[str, Any]: store = self._store() try: - run = store.run(run_id); attempt = store.active_attempt(run_id) + run, attempt = store.status_snapshot(run_id) return {"run_id": run_id, "state": run["state"], "phase": run["phase"], "version": run["version"], "active_attempt": {k: attempt.get(k) for k in ("id","status","pid","pgid","heartbeat_at")} if attempt else None, "next_action": "recovery_file_required" if run["state"] == "blocked" else None} finally: store.close() diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 772188a..43fa0a7 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import ContractError -SUPPORTED_SCHEMA_VERSION = 2 +SUPPORTED_SCHEMA_VERSION = 3 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} @@ -104,7 +104,10 @@ def migrate(self) -> None: raise SchemaVersionError(f"database schema {current} is newer than supported {SUPPORTED_SCHEMA_VERSION}") while current < SUPPORTED_SCHEMA_VERSION: next_version = current + 1 - sql = files("devsquad.migrations").joinpath(f"{next_version:03d}_" + ("initial.sql" if next_version == 1 else "supervisor.sql")).read_text() + candidates = [entry for entry in files("devsquad.migrations").iterdir() if entry.name.startswith(f"{next_version:03d}_") and entry.name.endswith(".sql")] + if len(candidates) != 1: + raise SchemaVersionError(f"migration {next_version} is missing or ambiguous") + sql = candidates[0].read_text() for statement in sql.split(";"): if statement.strip(): self.connection.execute(statement) @@ -180,6 +183,26 @@ def complete_preparation(self, run_id: str, fencing_token: int, mutable_snapshot self.connection.execute("ROLLBACK") raise + def fail_preparation(self, run_id: str, fencing_token: int, error: Any) -> int: + encoded = canonical_json(error) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,c.fencing_token,c.active FROM runs r JOIN claims c ON c.run_id=r.id WHERE r.id=?", + (run_id,), + ).fetchone() + if not row or row["state"] != "queued" or row["phase"] != "preparing" or not row["active"] or row["fencing_token"] != fencing_token: + raise ConflictError("preparation failure is stale or cancelled") + version, now = row["version"] + 1, _utc_now() + self.connection.execute("UPDATE runs SET state='failed',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("UPDATE claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.failed',?,?)", (run_id, version, encoded, now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + def cancel_preparing(self, run_id: str) -> int: self.connection.execute("BEGIN IMMEDIATE") try: @@ -316,7 +339,7 @@ def reserve_attempt(self, run_id: str, expected_version: int, owner_id: str, pac self.connection.execute("ROLLBACK") raise - def mark_attempt_running(self, reservation: AttemptReservation, pid: int, pgid: int, process_start_id: str) -> int: + def mark_attempt_running(self, reservation: AttemptReservation, pid: int, pgid: int, process_start_id: str, durable_paths: dict[str, str] | None = None) -> int: if any(type(value) is not int or value <= 0 for value in (pid, pgid)) or not process_start_id: raise ContractError("valid process identity is required") self.connection.execute("BEGIN IMMEDIATE") @@ -325,7 +348,8 @@ def mark_attempt_running(self, reservation: AttemptReservation, pid: int, pgid: if not row or row["version"] != reservation.version or row["phase"] != "launching" or row["status"] != "reserved" or row["attempt_token"] != reservation.attempt_token or row["fencing_token"] != reservation.supervisor_token or not row["active"]: raise ConflictError("attempt reservation is stale") version, now = row["version"] + 1, _utc_now() - self.connection.execute("UPDATE attempts SET status='running',pid=?,pgid=?,process_start_id=?,heartbeat_at=? WHERE id=?", (pid, pgid, process_start_id, now, reservation.attempt_id)) + durable_paths = durable_paths or {} + self.connection.execute("UPDATE attempts SET status='running',pid=?,pgid=?,process_start_id=?,heartbeat_at=?,stdout_spool=?,stderr_spool=?,stdout_meta=?,stderr_meta=?,exit_record=?,child_record=? WHERE id=?", (pid, pgid, process_start_id, now, durable_paths.get("stdout_spool"), durable_paths.get("stderr_spool"), durable_paths.get("stdout_meta"), durable_paths.get("stderr_meta"), durable_paths.get("exit_record"), durable_paths.get("child_record"), reservation.attempt_id)) self.connection.execute("UPDATE runs SET state='running',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, reservation.run_id)) event = canonical_json({"attempt_id": reservation.attempt_id, "pid": pid, "pgid": pgid, "process_start_id": process_start_id}) self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.running',?,?)", (reservation.run_id, version, event, now)) @@ -484,7 +508,21 @@ def events_page(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str raise ContractError("run does not exist") rows = self.connection.execute("SELECT id,run_version,type,payload,created_at FROM events WHERE run_id=? AND id>? ORDER BY id LIMIT ?", (run_id, after, limit + 1)).fetchall() page, more = rows[:limit], len(rows) > limit - return {"events": [{**dict(row), "payload": json.loads(row["payload"])} for row in page], "next_cursor": page[-1]["id"] if more and page else None} + consumed = page[-1]["id"] if page else after + return {"events": [{**dict(row), "payload": json.loads(row["payload"])} for row in page], "next_cursor": consumed, "has_more": more} + + def status_snapshot(self, run_id: str) -> tuple[dict[str, Any], dict[str, Any] | None]: + self.connection.execute("BEGIN") + try: + run = self.connection.execute("SELECT * FROM runs WHERE id=?", (run_id,)).fetchone() + if not run: + raise ContractError("run does not exist") + attempt = self.connection.execute("SELECT * FROM attempts WHERE run_id=? ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + self.connection.execute("COMMIT") + return dict(run), dict(attempt) if attempt else None + except Exception: + self.connection.execute("ROLLBACK") + raise def artifacts_for_run(self, run_id: str) -> list[dict[str, Any]]: return [dict(row) for row in self.connection.execute("SELECT id,name,path,sha256,byte_size,created_at FROM artifacts WHERE run_id=? ORDER BY created_at,id", (run_id,))] diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index b7bea59..1c12460 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -12,6 +12,8 @@ import threading import time from typing import Any, BinaryIO +from pathlib import Path +import json from .contracts import ContractError, LaunchSpec from .store import AttemptReservation, ConflictError, Store @@ -112,6 +114,13 @@ class RunningAttempt: stderr: BoundedDrain +@dataclass +class DurableAttempt: + reservation: AttemptReservation + process: subprocess.Popen[bytes] + paths: dict[str, str] + + class Supervisor: def __init__(self, store: Store, *, output_limit: int = 1024 * 1024, grace_seconds: float = 5.0): if output_limit <= 0 or grace_seconds < 0: @@ -165,6 +174,69 @@ def launch(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: stdout.start(); stderr.start() return RunningAttempt(reservation, process, started, stdout, stderr) + def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: str, package_digest: str) -> DurableAttempt: + reservation = self.store.reserve_attempt(run_id, expected_version, owner_id, package_digest) + directory = self.store.artifacts / run_id / f".{reservation.attempt_id}.spool" + directory.mkdir(parents=True, exist_ok=False) + paths = {name: str(directory / filename) for name, filename in { + "stdout_spool":"stdout.capture", "stderr_spool":"stderr.capture", "stdout_meta":"stdout.meta.json", "stderr_meta":"stderr.meta.json", "exit_record":"exit.json", "child_record":"child.json"}.items()} + gate_read, gate_write = os.pipe() + process = None + try: + command=[sys.executable,"-m","devsquad.attempt_runner","--gate-fd",str(gate_read), + "--database",str(self.store.database),"--artifacts",str(self.store.artifacts), + "--run-id",run_id,"--attempt-token",reservation.attempt_token, + "--supervisor-token",str(reservation.supervisor_token),"--stdout",paths["stdout_spool"], + "--stderr",paths["stderr_spool"],"--exit-record",paths["exit_record"], + "--child-record",paths["child_record"],"--limit",str(self.output_limit),"--",*spec.argv] + environment=os.environ.copy(); environment.update(spec.environment) + process=subprocess.Popen(command,cwd=spec.cwd,env=environment,stdin=subprocess.DEVNULL,stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL,pass_fds=(gate_read,),start_new_session=True) + os.close(gate_read) + started=process_start_identity(process.pid) + if started is None: raise RuntimeError("gated child has no strong process identity") + self.store.mark_attempt_running(reservation,process.pid,process.pid,started,paths) + os.write(gate_write,b"1"); os.close(gate_write) + return DurableAttempt(reservation,process,paths) + except Exception: + try: os.close(gate_write) + except OSError: pass + for fd in (gate_read,): + try: os.close(fd) + except OSError: pass + if process and process.poll() is None: + try: os.killpg(process.pid,signal.SIGKILL) + except ProcessLookupError: pass + process.wait(timeout=2) + self.store.fail_launch(reservation,"durable gated launch failed") + raise + + def wait_durable(self, handle: DurableAttempt, timeout_seconds: float) -> int: + deadline=time.monotonic()+timeout_seconds + self.grace_seconds + 5 + while handle.process.poll() is None and time.monotonic() None: + attempt=self.store.active_attempt(handle.reservation.run_id) + if not attempt or inspect_process(attempt["pid"],attempt["pgid"],attempt["process_start_id"])!="live": + raise ConflictError("durable process identity is unsafe to signal") + os.killpg(attempt["pgid"],signal.SIGTERM) + deadline=time.monotonic()+self.grace_seconds + while handle.process.poll() is None and time.monotonic() dict[str, Any]: stdout_meta, stderr_meta = handle.stdout.finish(), handle.stderr.finish() run_id, token = handle.reservation.run_id, handle.reservation.attempt_token diff --git a/test/core/test_service.py b/test/core/test_service.py index 9e2ba65..4de8dab 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -42,8 +42,8 @@ def wait_state(self, run_id, states, timeout=8): self.fail(f"run did not reach {states}: {self.service.status(run_id)}") def test_start_is_idempotent_and_result_events_are_durable(self): - first=self.service.start(self.task,"same") - second=self.service.start(self.task,"same") + first=self.service.start(self.task,"same",_internal_fake_delay=.01) + second=self.service.start(self.task,"same",_internal_fake_delay=.01) self.assertEqual(first["run_id"],second["run_id"]); self.assertFalse(second["created"]) self.wait_state(first["run_id"],{"succeeded"}) result=self.service.result(first["run_id"]) @@ -61,7 +61,7 @@ def test_new_process_inspects_run_after_launching_process_exits(self): status_command=[sys.executable,"-m","devsquad.cli","status",run_id,"--runtime-dir",str(self.runtime),"--json"] observed=subprocess.run(status_command,text=True,capture_output=True,env=env,check=True) self.assertEqual(json.loads(observed.stdout)["data"]["run_id"],run_id) - self.wait_state(run_id,{"succeeded"}) + self.assertEqual(self.service.status(run_id)["state"],"failed") def test_enqueue_crash_resumes_once_and_cancel_is_prompt(self): with mock.patch.object(self.service,"_spawn_daemon",return_value=0): diff --git a/test/core/test_store.py b/test/core/test_store.py index 70cdb9e..1152316 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -109,14 +109,14 @@ def test_artifact_is_finalized_and_verified_before_reference(self): self.assertEqual(row[1], 13) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 2) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(3,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 3) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(4,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) self.store = sqlite3.connect(":memory:") # tearDown-compatible close - def test_version_one_fixture_migrates_to_version_two(self): + def test_version_one_fixture_migrates_to_current(self): old_db = self.root / "old.sqlite3" connection = sqlite3.connect(old_db) sql = (ROOT / "plugin/core/src/devsquad/migrations/001_initial.sql").read_text() @@ -125,7 +125,7 @@ def test_version_one_fixture_migrates_to_version_two(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 2) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 3) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) From f76df4113549754dfc0b8f9660584f6083f7c6eb Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 9 Sep 2026 15:33:45 +0530 Subject: [PATCH 032/197] WIP checkpoint: WIP checkpoint: preserve M2 service recovery candidate (2026-09-09 15:33) --- docs/plans/engineering-team/M2-STATUS.md | 25 ++++-- docs/plans/engineering-team/RESUME.md | 9 +- plugin/core/src/devsquad/attempt_runner.py | 42 ++++++--- plugin/core/src/devsquad/detached.py | 5 +- .../devsquad/migrations/004_run_snapshot.sql | 3 + plugin/core/src/devsquad/service.py | 86 +++++++++++++++---- plugin/core/src/devsquad/store.py | 42 ++++++++- plugin/core/src/devsquad/supervisor.py | 61 ++++++++++--- plugin/core/src/devsquad/worker_gate.py | 11 +++ test/core/test_service.py | 29 ++++++- test/core/test_store.py | 17 +++- 11 files changed, 271 insertions(+), 59 deletions(-) create mode 100644 plugin/core/src/devsquad/migrations/004_run_snapshot.sql create mode 100644 plugin/core/src/devsquad/worker_gate.py diff --git a/docs/plans/engineering-team/M2-STATUS.md b/docs/plans/engineering-team/M2-STATUS.md index 202e7d7..55e7bf4 100644 --- a/docs/plans/engineering-team/M2-STATUS.md +++ b/docs/plans/engineering-team/M2-STATUS.md @@ -1,8 +1,8 @@ # M2 implementation status M2 is **in progress**. The store checkpoint is `0c2929c`; the bounded -supervisor lifecycle checkpoints are `0ee4cc0` and `df955f4`. Public detached service -operations and end-to-end shell persistence have not started. +supervisor lifecycle checkpoints are `0ee4cc0` and `df955f4`; the first +service checkpoint is `a0794a9`. | Store requirement | Evidence | Status | |---|---|---| @@ -17,12 +17,19 @@ operations and end-to-end shell persistence have not started. | Strong process identity and recovery | Darwin start second+microsecond identity is stable; live children remain owned without relaunch; dead and reused identities receive distinct recovery dispositions and reused IDs are never signalled | verified offline | | Bounded process lifecycle | Direct argv runs in a new session; heartbeat, PID/PGID/start identity, token and package digest persist; stdout/stderr drain continuously with truncation and full-stream hashes | verified offline | | Timeout and cancellation | Intent precedes verified TERM/KILL; TERM-resistant root and descendant disappear before completion; repeated terminal cancel is harmless | verified offline | +| Durable gated launch | A persisted attempt runner and an inner worker gate prevent task execution before strong runner and child identity records; the runner owns timeout, cancel polling, bounded spool files and an fsynced exit receipt | verified offline | +| Coordinator-loss recovery | A separate coordinator is killed while the runner lives; resume does not relaunch it, and a completed receipt is imported once with one attempt | verified offline | +| Frozen package | Every regular runtime asset is hashed, copied atomically, fsynced, stored on the run and verified before initial launch or resume | verified offline | +| Public workflow guard | Public branch-review tasks end with `CAPABILITY_UNAVAILABLE` until M3; only the private test argument can invoke the M2 fake step | verified offline | +| Result and event reads | Status is a transactional run/attempt snapshot; cursors always report the last consumed position and `has_more`; terminal service results verify referenced blob hashes and require a durable receipt for attempted runs | verified offline | +| Predecessor link | A superseded run must be terminal and belong to the same canonical Git project; the link is committed with preparation | verified offline | -Verification at this checkpoint: 62 core tests, including fourteen supervisor -tests and seven independent supervisor regressions; 10 legacy shell files with -202 assertions; and a temporary wheel installation that applied migration 2 -and imported the supervisor. +The recovery checkpoint `a0794a9` passed 67 core tests and 202 shell +assertions. The current service candidate passes 69 core tests, including a +coordinator-crash receipt-import case. Final shell and wheel evidence will be +recorded at the acceptance checkpoint. -The next slice is durable start/status/events/result/cancel/resume service -behavior and a supervisor process that survives the launching shell. This -checkpoint does not claim those behaviors or a working engineering workflow. +Remaining acceptance work is the independent service adversarial gate, +cross-process race expansion, CLI envelope verification and installed-wheel +migration 3 check. This is not a claim of a working engineering workflow; +branch-review execution begins in M3. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 8680031..8372ecf 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -25,7 +25,8 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne point are pending. - Cleanup-race hardening checkpoint: `df955f4`. Cleanup inventory fails closed, timeout/cancel identity races retain ownership fencing, and 62 core tests pass. -- Durable service checkpoint `cefd193` is explicit WIP. The following local +- Durable service checkpoint `cefd193` is explicit WIP and gated-runner + checkpoint `a0794a9` passed 67 core tests plus 202 shell assertions. The current local checkpoint replaces the unsafe anonymous-pipe launch with a persisted, gated attempt runner and migration 003 durable paths. The runner cannot launch the internal fixture worker until its strong identity and spool paths @@ -33,8 +34,10 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne atomic exit record. Ordered migration discovery, transactional status, stable event cursors and fenced preparation failure are implemented. Recovery/import of a runner exit record, predecessor validation, package - pin verification, result-receipt barrier and the full cross-process crash - matrix remain open. Public branch-review execution is explicitly failed as + pin verification, and the result-receipt barrier are now implemented and + covered by 69 core tests. The independent adversarial service gate, + cross-process race expansion, CLI envelopes and wheel migration-3 check + remain open. Public branch-review execution is explicitly failed as `CAPABILITY_UNAVAILABLE` until M3; only the internal test hook can run the fake step. M2 remains in progress. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. diff --git a/plugin/core/src/devsquad/attempt_runner.py b/plugin/core/src/devsquad/attempt_runner.py index ff48f35..ae59ec3 100644 --- a/plugin/core/src/devsquad/attempt_runner.py +++ b/plugin/core/src/devsquad/attempt_runner.py @@ -2,10 +2,16 @@ import argparse, hashlib, json, os, signal, subprocess, threading, time from pathlib import Path from .store import Store -from .supervisor import process_start_identity +from .supervisor import process_start_identity, inspect_process, _live_group_exists def _atomic(path: Path, value): - temporary=path.with_name(path.name+f".tmp.{os.getpid()}"); temporary.write_text(json.dumps(value,sort_keys=True)+"\n"); os.replace(temporary,path) + temporary=path.with_name(path.name+f".tmp.{os.getpid()}") + with temporary.open("w") as stream: + stream.write(json.dumps(value,sort_keys=True)+"\n"); stream.flush(); os.fsync(stream.fileno()) + os.replace(temporary,path) + directory=os.open(path.parent,os.O_RDONLY) + try: os.fsync(directory) + finally: os.close(directory) def _drain(stream, capture: Path, limit: int, result: dict): digest=hashlib.sha256(); total=0; kept=0 @@ -16,37 +22,51 @@ def _drain(stream, capture: Path, limit: int, result: dict): total+=len(chunk); digest.update(chunk); remaining=limit-kept if remaining>0: output.write(chunk[:remaining]); kept+=min(remaining,len(chunk)) output.flush(); os.fsync(output.fileno()) - result.update(total_bytes=total,captured_bytes=kept,truncated=total>kept,full_sha256=digest.hexdigest()) + result.update(total_bytes=total,captured_bytes=kept,truncated=total>kept,full_sha256=digest.hexdigest(),captured_sha256=hashlib.sha256(capture.read_bytes()).hexdigest()) def main(argv=None): - p=argparse.ArgumentParser(); p.add_argument("--gate-fd",type=int,required=True); p.add_argument("--database",type=Path,required=True); p.add_argument("--artifacts",type=Path,required=True); p.add_argument("--run-id",required=True); p.add_argument("--attempt-token",required=True); p.add_argument("--supervisor-token",type=int,required=True); p.add_argument("--stdout",type=Path,required=True); p.add_argument("--stderr",type=Path,required=True); p.add_argument("--exit-record",type=Path,required=True); p.add_argument("--child-record",type=Path,required=True); p.add_argument("--limit",type=int,required=True); p.add_argument("command",nargs=argparse.REMAINDER) + p=argparse.ArgumentParser(); p.add_argument("--gate-fd",type=int,required=True); p.add_argument("--database",type=Path,required=True); p.add_argument("--artifacts",type=Path,required=True); p.add_argument("--run-id",required=True); p.add_argument("--attempt-token",required=True); p.add_argument("--supervisor-token",type=int,required=True); p.add_argument("--stdout",type=Path,required=True); p.add_argument("--stderr",type=Path,required=True); p.add_argument("--exit-record",type=Path,required=True); p.add_argument("--child-record",type=Path,required=True); p.add_argument("--limit",type=int,required=True); p.add_argument("--timeout",type=float,required=True); p.add_argument("--grace",type=float,required=True); p.add_argument("command",nargs=argparse.REMAINDER) a=p.parse_args(argv); command=a.command[1:] if a.command[:1]==["--"] else a.command with os.fdopen(a.gate_fd,"rb",closefd=True) as gate: if gate.read(1)!=b"1": return 125 - child=subprocess.Popen(command,stdout=subprocess.PIPE,stderr=subprocess.PIPE,start_new_session=True) + child_gate_read,child_gate_write=os.pipe() + gated=[os.sys.executable,"-m","devsquad.worker_gate","--gate-fd",str(child_gate_read),"--",*command] + child=subprocess.Popen(gated,stdout=subprocess.PIPE,stderr=subprocess.PIPE,start_new_session=True,pass_fds=(child_gate_read,)) + os.close(child_gate_read) started=process_start_identity(child.pid) if started is None: child.kill(); child.wait(); return 126 _atomic(a.child_record,{"pid":child.pid,"pgid":child.pid,"process_start_id":started}) + os.write(child_gate_write,b"1"); os.close(child_gate_write) out,err={},{}; threads=[threading.Thread(target=_drain,args=(child.stdout,a.stdout,a.limit,out)),threading.Thread(target=_drain,args=(child.stderr,a.stderr,a.limit,err))] for thread in threads: thread.start() - store=Store(a.database,a.artifacts); cancelled=False + store=Store(a.database,a.artifacts); cancelled=False; timed_out=False; deadline=time.monotonic()+a.timeout try: while child.poll() is None: run=store.run(a.run_id) - if run["state"]=="cancelling": - cancelled=True + if run["state"]=="cancelling" or time.monotonic()>=deadline: + cancelled=run["state"]=="cancelling"; timed_out=not cancelled + if inspect_process(child.pid,child.pid,started)!="live": raise RuntimeError("child identity became unsafe") try: os.killpg(child.pid,signal.SIGTERM) except ProcessLookupError: pass - try: child.wait(timeout=5) + try: child.wait(timeout=a.grace) except subprocess.TimeoutExpired: try: os.killpg(child.pid,signal.SIGKILL) except ProcessLookupError: pass + child.wait() break store.heartbeat_attempt(a.run_id,a.attempt_token,a.supervisor_token); time.sleep(.1) returncode=child.wait() finally: store.close() - for thread in threads: thread.join() - _atomic(a.exit_record,{"returncode":returncode,"cancelled":cancelled,"stdout":out,"stderr":err,"finished_at":time.time()}) + if _live_group_exists(child.pid): + try: os.killpg(child.pid,signal.SIGKILL) + except ProcessLookupError: pass + cleanup_deadline=time.monotonic()+a.grace + while _live_group_exists(child.pid) and time.monotonic() Store: def _freeze_package(self) -> tuple[Path, str]: source = Path(__file__).resolve().parent digest = hashlib.sha256() - members = sorted(p for p in source.rglob("*") if p.is_file() and "__pycache__" not in p.parts and p.suffix in {".py", ".sql"}) + members = sorted(p for p in source.rglob("*") if p.is_file() and "__pycache__" not in p.parts and p.suffix != ".pyc") for path in members: relative = path.relative_to(source) digest.update(str(relative).encode() + b"\0" + path.read_bytes()) @@ -40,12 +40,39 @@ def _freeze_package(self) -> tuple[Path, str]: temporary = Path(tempfile.mkdtemp(prefix=f".{value}.", dir=destination.parent.parent)) try: shutil.copytree(source, temporary / "devsquad", ignore=shutil.ignore_patterns("__pycache__", "*.pyc")) - destination.parent.mkdir(parents=True, exist_ok=True) - os.replace(temporary / "devsquad", destination) + for copied in (temporary/"devsquad").rglob("*"): + if copied.is_file(): + with copied.open("rb") as stream: os.fsync(stream.fileno()) + try: + os.rename(temporary, destination.parent) + except OSError: + if not destination.exists(): raise finally: shutil.rmtree(temporary, ignore_errors=True) + if self._package_digest(destination.parent)!=value: + raise ConflictError("frozen package cache is corrupt") + for directory_path in (destination,destination.parent,destination.parent.parent): + descriptor=os.open(directory_path,os.O_RDONLY) + try: os.fsync(descriptor) + finally: os.close(descriptor) return destination.parent, value + @staticmethod + def _package_digest(package: Path) -> str: + source=package/"devsquad"; digest=hashlib.sha256() + for path in sorted(p for p in source.rglob("*") if p.is_file() and "__pycache__" not in p.parts and p.suffix != ".pyc"): + relative=path.relative_to(source); digest.update(str(relative).encode()+b"\0"+path.read_bytes()) + return digest.hexdigest() + + def _verified_package(self, run: dict[str, Any]) -> tuple[Path,str]: + if not run.get("package_path") or not run.get("package_digest"): + raise ConflictError("run has no pinned package") + package=Path(run["package_path"]).resolve() + expected_root=(self.runtime/"packages").resolve() + if expected_root not in package.parents or self._package_digest(package)!=run["package_digest"]: + raise ConflictError("pinned package is missing or corrupt") + return package,run["package_digest"] + @staticmethod def _resolve_snapshot(task: dict[str, Any], internal_delay: float | None) -> dict[str, Any]: repo = Path(task["project"]["repo_path"]).resolve(strict=True) @@ -75,30 +102,33 @@ def start(self, task: dict[str, Any], idempotency_key: str, supersedes_run_id: s return {"run_id": claim.run_id, "state": store.run(claim.run_id)["state"], "created": False} try: snapshot = self._resolve_snapshot(task, _internal_fake_delay) + if _internal_fake_delay is None: + store.fail_preparation(claim.run_id, claim.fencing_token or 0, {"error": "CAPABILITY_UNAVAILABLE", "message": "branch-review workflow is introduced in M3"}) + return {"run_id": claim.run_id, "state": "failed", "created": True} + package, digest = self._freeze_package() + version = store.complete_preparation(claim.run_id, claim.fencing_token or 0, snapshot, package_path=str(package), package_digest=digest, supersedes_run_id=supersedes_run_id) except Exception as exc: store.fail_preparation(claim.run_id, claim.fencing_token or 0, {"error": "PREPARATION_FAILED", "message": str(exc)}) raise - if _internal_fake_delay is None: - store.fail_preparation(claim.run_id, claim.fencing_token or 0, {"error": "CAPABILITY_UNAVAILABLE", "message": "branch-review workflow is introduced in M3"}) - return {"run_id": claim.run_id, "state": "failed", "created": True} - version = store.complete_preparation(claim.run_id, claim.fencing_token or 0, snapshot) finally: store.close() - package, digest = self._freeze_package() self._spawn_daemon(claim.run_id, version, package, digest) return {"run_id": claim.run_id, "state": "queued", "created": True} def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, digest: str) -> int: command = [sys.executable, "-m", "devsquad.detached", "--database", str(self.database), "--artifacts", str(self.artifacts), "--run-id", run_id, "--expected-version", str(expected_version), "--package-digest", digest] environment = {"PATH": os.environ.get("PATH", ""), "PYTHONPATH": str(package)} - process = subprocess.Popen(command, cwd=self.runtime, env=environment, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, start_new_session=True, close_fds=True) + log_dir=self.runtime/"private-logs"; log_dir.mkdir(parents=True,exist_ok=True) + with (log_dir/f"{run_id}.supervisor.log").open("ab",buffering=0) as diagnostic: + process = subprocess.Popen(command, cwd=self.runtime, env=environment, stdin=subprocess.DEVNULL, stdout=diagnostic, stderr=diagnostic, start_new_session=True, close_fds=True) return process.pid def status(self, run_id: str) -> dict[str, Any]: store = self._store() try: run, attempt = store.status_snapshot(run_id) - return {"run_id": run_id, "state": run["state"], "phase": run["phase"], "version": run["version"], "active_attempt": {k: attempt.get(k) for k in ("id","status","pid","pgid","heartbeat_at")} if attempt else None, "next_action": "recovery_file_required" if run["state"] == "blocked" else None} + active=attempt if attempt and attempt.get("status") in {"reserved","running","cancelling","ownership_ambiguous"} else None + return {"run_id": run_id, "state": run["state"], "phase": run["phase"], "version": run["version"], "active_attempt": {k: active.get(k) for k in ("id","status","pid","pgid","heartbeat_at")} if active else None, "next_action": "recovery_file_required" if run["state"] == "blocked" else None} finally: store.close() def events(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any]: @@ -109,15 +139,26 @@ def events(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any def result(self, run_id: str) -> dict[str, Any]: store = self._store() try: - run = store.run(run_id) - return {"run_id": run_id, "ready": run["state"] in TERMINAL_STATES, "state": run["state"], "artifacts": store.artifacts_for_run(run_id) if run["state"] in TERMINAL_STATES else []} + run, artifacts = store.result_snapshot(run_id) + if run["state"] not in TERMINAL_STATES: + return {"run_id":run_id,"ready":False,"state":run["state"],"artifacts":[]} + if run["state"] != "failed" or run.get("package_digest"): + if not any(item["name"]=="result-receipt.json" for item in artifacts): + raise ConflictError("terminal result has no durable receipt") + for item in artifacts: + path=Path(item["path"]) + if not path.is_file() or hashlib.sha256(path.read_bytes()).hexdigest()!=item["sha256"]: + raise ConflictError("referenced result artifact is missing or corrupt") + return {"run_id":run_id,"ready":True,"state":run["state"],"artifacts":artifacts} finally: store.close() def cancel(self, run_id: str) -> dict[str, Any]: store = self._store() try: run = store.run(run_id) - if run["state"] == "queued" and run["phase"] is None: version = store.cancel_queued(run_id) + if run["state"] == "queued" and run["phase"] == "preparing": version = store.cancel_preparing(run_id) + elif run["state"] == "queued" and run["phase"] == "launching": version = store.cancel_launching(run_id) + elif run["state"] == "queued" and run["phase"] is None: version = store.cancel_queued(run_id) elif run["state"] in {"running", "cancelling"}: version, _ = store.request_cancel(run_id) elif run["state"] in TERMINAL_STATES: version = run["version"] else: raise ConflictError("run requires recovery before cancellation") @@ -129,14 +170,25 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st try: run = store.run(run_id) if run["state"] in TERMINAL_STATES: raise ConflictError("terminal run cannot resume; start a superseding run") - if run["state"] == "running": + if run["state"] in {"running","cancelling"}: from .supervisor import Supervisor - disposition = Supervisor(store).recover(run_id) + attempt=store.attempt(run_id) + disposition = Supervisor(store).import_durable(run_id) if attempt and attempt.get("exit_record") else Supervisor(store).recover(run_id) return {"run_id": run_id, "disposition": disposition, "launched": False} if run["state"] == "queued" and run["phase"] is None: version = run["version"] + elif run["state"] == "blocked": + if not isinstance(recovery,dict) or set(recovery)!={"attempt_id","disposition"} or recovery["disposition"] not in {"confirm_dead","retain_ownership"}: + raise ContractError("blocked resume requires a typed recovery disposition") + attempt=store.attempt(run_id) + if not attempt or recovery["attempt_id"]!=attempt["id"]: + raise ConflictError("recovery disposition targets a different attempt") + required="retain_ownership" if attempt["status"]=="ownership_ambiguous" else "confirm_dead" + if recovery["disposition"]!=required: + raise ConflictError("recovery disposition contradicts persisted ownership") + return {"run_id":run_id,"disposition":required,"launched":False} else: - raise ConflictError("run requires an explicit recovery disposition") + raise ConflictError("run is not resumable") finally: store.close() - package, digest = self._freeze_package(); self._spawn_daemon(run_id, version, package, digest) + package,digest=self._verified_package(run); self._spawn_daemon(run_id, version, package, digest) return {"run_id": run_id, "disposition": "continued", "launched": True} diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 43fa0a7..38d9330 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import ContractError -SUPPORTED_SCHEMA_VERSION = 3 +SUPPORTED_SCHEMA_VERSION = 4 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} @@ -163,7 +163,7 @@ def claim_start(self, worktree: Path, idempotency_key: str, submitted_request: A self.connection.execute("ROLLBACK") raise - def complete_preparation(self, run_id: str, fencing_token: int, mutable_snapshot: Any) -> int: + def complete_preparation(self, run_id: str, fencing_token: int, mutable_snapshot: Any, *, package_path: str | None = None, package_digest: str | None = None, supersedes_run_id: str | None = None) -> int: snapshot = canonical_json(mutable_snapshot) self.connection.execute("BEGIN IMMEDIATE") try: @@ -173,8 +173,13 @@ def complete_preparation(self, run_id: str, fencing_token: int, mutable_snapshot ).fetchone() if not row or row["state"] != "queued" or row["phase"] != "preparing" or not row["active"] or row["fencing_token"] != fencing_token: raise ConflictError("preparation claim is stale or cancelled") + if supersedes_run_id is not None: + predecessor = self.connection.execute("SELECT project_id,state FROM runs WHERE id=?", (supersedes_run_id,)).fetchone() + project = self.connection.execute("SELECT project_id FROM runs WHERE id=?", (run_id,)).fetchone() + if not predecessor or predecessor["project_id"] != project["project_id"] or predecessor["state"] not in TERMINAL_STATES: + raise ConflictError("superseded run must be terminal and belong to the same project") version, now = row["version"] + 1, _utc_now() - self.connection.execute("UPDATE runs SET mutable_snapshot=?,phase=NULL,version=?,updated_at=? WHERE id=?", (snapshot, version, now, run_id)) + self.connection.execute("UPDATE runs SET mutable_snapshot=?,package_path=?,package_digest=?,supersedes_run_id=?,phase=NULL,version=?,updated_at=? WHERE id=?", (snapshot, package_path, package_digest, supersedes_run_id, version, now, run_id)) self.connection.execute("UPDATE claims SET active=0 WHERE run_id=?", (run_id,)) self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.queued','{}',?)", (run_id, version, now)) self.connection.execute("COMMIT") @@ -416,6 +421,7 @@ def finish_attempt(self, run_id: str, attempt_token: str, terminal_state: str, p version, now = run["version"] + 1, _utc_now() self.connection.execute("UPDATE attempts SET status='finished',finished_at=? WHERE id=?", (now, attempt["id"])) self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + terminal_state = "cancelled" if run["state"] == "cancelling" else terminal_state self.connection.execute("UPDATE runs SET state=?,phase=NULL,version=?,updated_at=? WHERE id=?", (terminal_state, version, now, run_id)) self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", (run_id, version, f"run.{terminal_state}", encoded, now)) self.connection.execute("COMMIT") @@ -501,6 +507,19 @@ def cancel_queued(self, run_id: str) -> int: except Exception: self.connection.execute("ROLLBACK"); raise + def cancel_launching(self, run_id: str) -> int: + self.connection.execute("BEGIN IMMEDIATE") + try: + row=self.connection.execute("SELECT state,phase,version FROM runs WHERE id=?",(run_id,)).fetchone() + if not row or row["state"]!="queued" or row["phase"]!="launching": raise ConflictError("run is not launching") + version,now=row["version"]+1,_utc_now() + self.connection.execute("UPDATE attempts SET status='recovery_required',finished_at=? WHERE run_id=? AND status='reserved'",(now,run_id)) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?",(run_id,)) + self.connection.execute("UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?",(version,now,run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelled','{}',?)",(run_id,version,now)) + self.connection.execute("COMMIT"); return version + except Exception: self.connection.execute("ROLLBACK"); raise + def events_page(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any]: if type(after) is not int or after < 0 or type(limit) is not int or not 1 <= limit <= 1000: raise ContractError("event cursor/limit is invalid") @@ -524,9 +543,26 @@ def status_snapshot(self, run_id: str) -> tuple[dict[str, Any], dict[str, Any] | self.connection.execute("ROLLBACK") raise + def result_snapshot(self, run_id: str) -> tuple[dict[str, Any], list[dict[str, Any]]]: + self.connection.execute("BEGIN") + try: + run=self.connection.execute("SELECT * FROM runs WHERE id=?",(run_id,)).fetchone() + if not run: raise ContractError("run does not exist") + artifacts=self.connection.execute("SELECT id,name,path,sha256,byte_size,created_at FROM artifacts WHERE run_id=? ORDER BY created_at,id",(run_id,)).fetchall() + self.connection.execute("COMMIT"); return dict(run),[dict(row) for row in artifacts] + except Exception: self.connection.execute("ROLLBACK"); raise + def artifacts_for_run(self, run_id: str) -> list[dict[str, Any]]: return [dict(row) for row in self.connection.execute("SELECT id,name,path,sha256,byte_size,created_at FROM artifacts WHERE run_id=? ORDER BY created_at,id", (run_id,))] + def artifact_named(self, run_id: str, name: str) -> dict[str, Any] | None: + row=self.connection.execute("SELECT id,name,path,sha256,byte_size,created_at FROM artifacts WHERE run_id=? AND name=?",(run_id,name)).fetchone() + return dict(row) if row else None + + def attempt(self, run_id: str) -> dict[str, Any] | None: + row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + return dict(row) if row else None + def run(self, run_id: str) -> dict[str, Any]: row = self.connection.execute("SELECT * FROM runs WHERE id=?", (run_id,)).fetchone() if not row: raise ContractError("run does not exist") diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 1c12460..a29959e 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -16,7 +16,7 @@ import json from .contracts import ContractError, LaunchSpec -from .store import AttemptReservation, ConflictError, Store +from .store import AttemptReservation, ConflictError, Store, canonical_json def process_start_identity(pid: int) -> str | None: @@ -188,7 +188,8 @@ def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, o "--run-id",run_id,"--attempt-token",reservation.attempt_token, "--supervisor-token",str(reservation.supervisor_token),"--stdout",paths["stdout_spool"], "--stderr",paths["stderr_spool"],"--exit-record",paths["exit_record"], - "--child-record",paths["child_record"],"--limit",str(self.output_limit),"--",*spec.argv] + "--child-record",paths["child_record"],"--limit",str(self.output_limit), + "--timeout",str(spec.timeout_seconds),"--grace",str(self.grace_seconds),"--",*spec.argv] environment=os.environ.copy(); environment.update(spec.environment) process=subprocess.Popen(command,cwd=spec.cwd,env=environment,stdin=subprocess.DEVNULL,stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL,pass_fds=(gate_read,),start_new_session=True) os.close(gate_read) @@ -218,16 +219,56 @@ def wait_durable(self, handle: DurableAttempt, timeout_seconds: float) -> int: self.store.request_cancel(handle.reservation.run_id) handle.process.wait(timeout=self.grace_seconds+5) returncode=handle.process.wait(timeout=2) - receipt=json.loads(Path(handle.paths["exit_record"]).read_text()) - metadata={"stdout":receipt["stdout"],"stderr":receipt["stderr"]} - run_id=handle.reservation.run_id; token=handle.reservation.attempt_token - stdout_id=self.store.store_artifact(run_id,f"{handle.reservation.attempt_id}.stdout",Path(handle.paths["stdout_spool"]).read_bytes()) - stderr_id=self.store.store_artifact(run_id,f"{handle.reservation.attempt_id}.stderr",Path(handle.paths["stderr_spool"]).read_bytes()) - self.store.record_attempt_output(run_id,token,stdout_id,stderr_id,metadata) - terminal="cancelled" if receipt["cancelled"] else ("succeeded" if returncode==0 else "failed") - self.store.finish_attempt(run_id,token,terminal,{"returncode":returncode,"output":metadata}) + self.import_durable(handle.reservation.run_id) return returncode + def import_durable(self, run_id: str) -> str: + attempt=self.store.attempt(run_id) + if not attempt or attempt["status"] not in {"running","cancelling"}: + return "already_finalized" + identity=inspect_process(attempt["pid"],attempt["pgid"],attempt["process_start_id"]) + if identity=="live": return "live" + receipt_path=Path(attempt["exit_record"] or "") + if identity=="ambiguous": + self.store.block_recovery(run_id,attempt["attempt_token"],"attempt runner identity is ambiguous") + return "ownership_ambiguous" + if not receipt_path.is_file(): + child_path=Path(attempt["child_record"] or "") + if child_path.is_file(): + try: + child=json.loads(child_path.read_text()) + if inspect_process(child["pid"],child["pgid"],child["process_start_id"])!="dead": + self.store.block_recovery(run_id,attempt["attempt_token"],"runner died while its child may still be live") + return "ownership_ambiguous" + except (OSError,ValueError,KeyError,TypeError): + self.store.block_recovery(run_id,attempt["attempt_token"],"child identity record is invalid") + return "ownership_ambiguous" + self.store.block_recovery(run_id,attempt["attempt_token"],"runner died without an exit receipt",release_writer=True) + return "recovery_required" + try: + receipt=json.loads(receipt_path.read_text()) + if type(receipt.get("returncode")) is not int or type(receipt.get("cancelled")) is not bool: + raise ValueError("invalid receipt") + metadata={name:receipt[name] for name in ("stdout","stderr")} + ids=[] + for stream,column in (("stdout","stdout_spool"),("stderr","stderr_spool")): + data=Path(attempt[column]).read_bytes(); meta=metadata[stream] + if hashlib.sha256(data).hexdigest()!=meta["captured_sha256"] or len(data)!=meta["captured_bytes"]: + raise ValueError("capture hash mismatch") + logical=f"{attempt['id']}.{stream}"; existing=self.store.artifact_named(run_id,logical) + ids.append(existing["id"] if existing else self.store.store_artifact(run_id,logical,data)) + if not attempt.get("output_metadata"): + self.store.record_attempt_output(run_id,attempt["attempt_token"],ids[0],ids[1],metadata) + receipt_bytes=canonical_json(receipt).encode() + if not self.store.artifact_named(run_id,"result-receipt.json"): + self.store.store_artifact(run_id,"result-receipt.json",receipt_bytes) + terminal="cancelled" if receipt["cancelled"] else ("succeeded" if receipt["returncode"]==0 else "failed") + self.store.finish_attempt(run_id,attempt["attempt_token"],terminal,{"returncode":receipt["returncode"],"receipt":"result-receipt.json"}) + return terminal + except (OSError,ValueError,KeyError,TypeError,json.JSONDecodeError): + self.store.block_recovery(run_id,attempt["attempt_token"],"durable receipt or capture is invalid") + return "ownership_ambiguous" + def _terminate_durable(self, handle: DurableAttempt) -> None: attempt=self.store.active_attempt(handle.reservation.run_id) if not attempt or inspect_process(attempt["pid"],attempt["pgid"],attempt["process_start_id"])!="live": diff --git a/plugin/core/src/devsquad/worker_gate.py b/plugin/core/src/devsquad/worker_gate.py new file mode 100644 index 0000000..7b24896 --- /dev/null +++ b/plugin/core/src/devsquad/worker_gate.py @@ -0,0 +1,11 @@ +"""Execute an internal worker only after its supervisor persists ownership.""" +import argparse, os + +def main(argv=None): + parser=argparse.ArgumentParser(); parser.add_argument("--gate-fd",type=int,required=True); parser.add_argument("command",nargs=argparse.REMAINDER) + args=parser.parse_args(argv); command=args.command[1:] if args.command[:1]==["--"] else args.command + with os.fdopen(args.gate_fd,"rb",closefd=True) as gate: + if gate.read(1)!=b"1": return 125 + os.execvpe(command[0],command,os.environ) + +if __name__=="__main__": raise SystemExit(main()) diff --git a/test/core/test_service.py b/test/core/test_service.py index 4de8dab..db79677 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -47,7 +47,7 @@ def test_start_is_idempotent_and_result_events_are_durable(self): self.assertEqual(first["run_id"],second["run_id"]); self.assertFalse(second["created"]) self.wait_state(first["run_id"],{"succeeded"}) result=self.service.result(first["run_id"]) - self.assertTrue(result["ready"]); self.assertEqual(len(result["artifacts"]),2) + self.assertTrue(result["ready"]); self.assertEqual(len(result["artifacts"]),3) page=self.service.events(first["run_id"],0,2) self.assertEqual(len(page["events"]),2); self.assertIsNotNone(page["next_cursor"]) with self.assertRaises(ConflictError): self.service.resume(first["run_id"]) @@ -83,7 +83,7 @@ def test_dead_supervisor_with_live_child_never_relaunches(self): daemon_pid=int(owner.split(":",1)[1]); os.kill(daemon_pid,signal.SIGKILL) time.sleep(.1) resumed=self.service.resume(started["run_id"]) - self.assertFalse(resumed["launched"]); self.assertEqual(resumed["disposition"],"live_owned") + self.assertFalse(resumed["launched"]); self.assertEqual(resumed["disposition"],"live") self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM attempts WHERE run_id=?",(started["run_id"],)).fetchone()[0],1) os.killpg(attempt["pgid"],signal.SIGKILL) finally: store.close() @@ -96,5 +96,30 @@ def test_orphan_artifact_is_not_a_result(self): finally: store.close() self.assertEqual(self.service.result(started["run_id"])["artifacts"],[]) + def test_coordinator_crash_imports_runner_receipt_once(self): + started=self.service.start(self.task,"receipt-recovery",_internal_fake_delay=.3) + self.wait_state(started["run_id"],{"running"}) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + owner=store.connection.execute("SELECT owner_id FROM supervisor_claims WHERE run_id=?",(started["run_id"],)).fetchone()[0] + os.kill(int(owner.split(":",1)[1]),signal.SIGKILL) + finally: store.close() + time.sleep(.6) + first=self.service.resume(started["run_id"]) + self.assertIn(first["disposition"],{"succeeded","already_finalized"}) + self.assertTrue(self.service.result(started["run_id"])["ready"]) + with self.assertRaises(ConflictError): self.service.resume(started["run_id"]) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM attempts WHERE run_id=?",(started["run_id"],)).fetchone()[0],1) + finally: store.close() + + def test_supersedes_requires_terminal_same_project(self): + predecessor=self.service.start(self.task,"predecessor") + replacement=self.service.start(self.task,"replacement",predecessor["run_id"],_internal_fake_delay=.01) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: self.assertEqual(store.run(replacement["run_id"])["supersedes_run_id"],predecessor["run_id"]) + finally: store.close() + self.wait_state(replacement["run_id"],{"succeeded"}) + if __name__=="__main__": unittest.main() diff --git a/test/core/test_store.py b/test/core/test_store.py index 1152316..86eaa05 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -109,8 +109,8 @@ def test_artifact_is_finalized_and_verified_before_reference(self): self.assertEqual(row[1], 13) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 3) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(4,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 4) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(5,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -125,9 +125,20 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 3) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 4) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) + def test_version_three_fixture_adds_run_snapshot_columns(self): + old_db=self.root/"v3.sqlite3"; connection=sqlite3.connect(old_db) + for version,name in ((1,"001_initial.sql"),(2,"002_supervisor.sql"),(3,"003_durable_io.sql")): + connection.executescript((ROOT/"plugin/core/src/devsquad/migrations"/name).read_text()) + connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) + connection.commit(); connection.close() + upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],4) + columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} + self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) + if __name__ == "__main__": unittest.main() From 5aa2e74f90be71caaf440abd2df6b9e86ad6766f Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 9 Sep 2026 21:57:29 +0530 Subject: [PATCH 033/197] fix: harden durable M2 receipt recovery --- plugin/core/src/devsquad/attempt_runner.py | 8 +- plugin/core/src/devsquad/detached.py | 2 +- plugin/core/src/devsquad/service.py | 2 +- plugin/core/src/devsquad/store.py | 112 +++++++++++++++++++++ plugin/core/src/devsquad/supervisor.py | 41 +++++--- test/core/test_m2_supervisor_gate.py | 27 +++++ test/core/test_service.py | 71 +++++++++++++ 7 files changed, 243 insertions(+), 20 deletions(-) diff --git a/plugin/core/src/devsquad/attempt_runner.py b/plugin/core/src/devsquad/attempt_runner.py index ae59ec3..e98ab02 100644 --- a/plugin/core/src/devsquad/attempt_runner.py +++ b/plugin/core/src/devsquad/attempt_runner.py @@ -30,7 +30,7 @@ def main(argv=None): with os.fdopen(a.gate_fd,"rb",closefd=True) as gate: if gate.read(1)!=b"1": return 125 child_gate_read,child_gate_write=os.pipe() - gated=[os.sys.executable,"-m","devsquad.worker_gate","--gate-fd",str(child_gate_read),"--",*command] + gated=[os.sys.executable,"-P","-m","devsquad.worker_gate","--gate-fd",str(child_gate_read),"--",*command] child=subprocess.Popen(gated,stdout=subprocess.PIPE,stderr=subprocess.PIPE,start_new_session=True,pass_fds=(child_gate_read,)) os.close(child_gate_read) started=process_start_identity(child.pid) @@ -67,6 +67,8 @@ def main(argv=None): for thread in threads: thread.join(timeout=a.grace+1) if any(thread.is_alive() for thread in threads): raise RuntimeError("durable output drain did not finish") child.stdout.close(); child.stderr.close() - _atomic(a.exit_record,{"returncode":returncode,"cancelled":cancelled,"timed_out":timed_out,"stdout":out,"stderr":err,"finished_at":time.time()}) - return returncode + receipt={"returncode":returncode,"cancelled":cancelled,"timed_out":timed_out,"stdout":out,"stderr":err,"finished_at":time.time()} + if timed_out: receipt["error"]="TIMEOUT" + _atomic(a.exit_record,receipt) + return 124 if timed_out else returncode if __name__=="__main__": raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index 60ade0a..1f89c0b 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -24,7 +24,7 @@ def main(argv=None): snapshot = json.loads(run["mutable_snapshot"]) environment = {"DEVSQUAD_WORKER": "1", "DEVSQUAD_RUN_ID": args.run_id} identity = ExecutionIdentity("devsquad-fake-step", "1", None, None, None, None) - command = [sys.executable, "-m", "devsquad.fake_step"] + command = [sys.executable, "-P", "-m", "devsquad.fake_step"] if "internal_fake_delay" in snapshot: command += ["--delay", str(snapshot["internal_fake_delay"])] spec = LaunchSpec(1, "devsquad-fake-step", "cli_exec", tuple(command), run["worktree_path"], None, snapshot["task"]["budget"]["wall_seconds"], identity, environment) supervisor = Supervisor(store) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 6165e45..1e2d5ec 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -116,7 +116,7 @@ def start(self, task: dict[str, Any], idempotency_key: str, supersedes_run_id: s return {"run_id": claim.run_id, "state": "queued", "created": True} def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, digest: str) -> int: - command = [sys.executable, "-m", "devsquad.detached", "--database", str(self.database), "--artifacts", str(self.artifacts), "--run-id", run_id, "--expected-version", str(expected_version), "--package-digest", digest] + command = [sys.executable, "-P", "-m", "devsquad.detached", "--database", str(self.database), "--artifacts", str(self.artifacts), "--run-id", run_id, "--expected-version", str(expected_version), "--package-digest", digest] environment = {"PATH": os.environ.get("PATH", ""), "PYTHONPATH": str(package)} log_dir=self.runtime/"private-logs"; log_dir.mkdir(parents=True,exist_ok=True) with (log_dir/f"{run_id}.supervisor.log").open("ab",buffering=0) as diagnostic: diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 38d9330..30be3e1 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -448,6 +448,118 @@ def record_attempt_output(self, run_id: str, attempt_token: str, stdout_id: str, self.connection.execute("ROLLBACK") raise + def commit_durable_import(self, run_id: str, attempt_token: str, artifacts: list[dict[str, Any]], metadata: Any, terminal_state: str, payload: Any) -> str: + """Atomically import one durable receipt, or observe its prior import. + + Content-addressed files are finalized before this call. All database + references, output projection fields, and terminal state then cross a + single write fence so competing recovery processes cannot partially + import or downgrade a valid completion. + """ + if terminal_state not in TERMINAL_STATES: + raise ContractError("invalid terminal state") + expected_parent = (self.artifacts / run_id).resolve() + prepared = [] + names = set() + for artifact in artifacts: + name = artifact.get("name") + path = Path(artifact.get("path", "")) + digest = artifact.get("sha256") + size = artifact.get("byte_size") + if not name or Path(name).name != name or name in names: + raise ContractError("durable artifact names must be unique path components") + names.add(name) + if path.resolve().parent != expected_parent: + raise ContractError("durable artifact path is outside the run-owned store") + content = path.read_bytes() + if hashlib.sha256(content).hexdigest() != digest or len(content) != size: + raise ConflictError("durable artifact changed before database import") + prepared.append((name, str(path), digest, size)) + stdout_name = next((name for name in names if name.endswith(".stdout")), None) + stderr_name = next((name for name in names if name.endswith(".stderr")), None) + if stdout_name is None or stderr_name is None or "result-receipt.json" not in names: + raise ContractError("durable import requires stdout, stderr, and result receipt artifacts") + encoded_metadata, encoded_payload = canonical_json(metadata), canonical_json(payload) + + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute("SELECT state,version FROM runs WHERE id=?", (run_id,)).fetchone() + attempt = self.connection.execute( + "SELECT id,status,stdout_artifact_id,stderr_artifact_id,output_metadata FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("durable import is fenced") + if attempt["status"] == "finished" and run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["state"] + if attempt["status"] not in {"running", "cancelling"} or run["state"] not in {"running", "cancelling"}: + raise ConflictError("durable import is fenced") + + version = run["version"] + artifact_ids = {} + for name, path, digest, size in prepared: + existing = self.connection.execute( + "SELECT id,path,sha256,byte_size FROM artifacts WHERE run_id=? AND name=?", (run_id, name), + ).fetchone() + if existing: + if existing["path"] != path or existing["sha256"] != digest or existing["byte_size"] != size: + raise ConflictError("durable artifact conflicts with an existing reference") + artifact_ids[name] = existing["id"] + continue + artifact_id, now = str(uuid.uuid4()), _utc_now() + self.connection.execute( + "INSERT INTO artifacts(id,run_id,name,path,sha256,byte_size,created_at) VALUES(?,?,?,?,?,?,?)", + (artifact_id, run_id, name, path, digest, size, now), + ) + artifact_ids[name] = artifact_id + version += 1 + event = canonical_json({"artifact_id": artifact_id, "name": name, "sha256": digest, "byte_size": size}) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'artifact.recorded',?,?)", + (run_id, version, event, now), + ) + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id), + ) + + stdout_id, stderr_id = artifact_ids[stdout_name], artifact_ids[stderr_name] + if attempt["output_metadata"] is None: + now = _utc_now(); version += 1 + self.connection.execute( + "UPDATE attempts SET stdout_artifact_id=?,stderr_artifact_id=?,output_metadata=? WHERE id=?", + (stdout_id, stderr_id, encoded_metadata, attempt["id"]), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'attempt.output',?,?)", + (run_id, version, encoded_metadata, now), + ) + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id), + ) + elif (attempt["stdout_artifact_id"] != stdout_id + or attempt["stderr_artifact_id"] != stderr_id + or attempt["output_metadata"] != encoded_metadata): + raise ConflictError("durable output conflicts with its prior import") + + now = _utc_now(); version += 1 + effective_state = "cancelled" if run["state"] == "cancelling" else terminal_state + self.connection.execute("UPDATE attempts SET status='finished',finished_at=? WHERE id=?", (now, attempt["id"])) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute( + "UPDATE runs SET state=?,phase=NULL,version=?,updated_at=? WHERE id=?", + (effective_state, version, now, run_id), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", + (run_id, version, f"run.{effective_state}", encoded_payload, now), + ) + self.connection.execute("COMMIT") + return effective_state + except Exception: + self.connection.execute("ROLLBACK") + raise + def block_recovery(self, run_id: str, attempt_token: str, reason: str, *, release_writer: bool = False) -> int: self.connection.execute("BEGIN IMMEDIATE") try: diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index a29959e..64bfbe8 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -183,7 +183,7 @@ def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, o gate_read, gate_write = os.pipe() process = None try: - command=[sys.executable,"-m","devsquad.attempt_runner","--gate-fd",str(gate_read), + command=[sys.executable,"-P","-m","devsquad.attempt_runner","--gate-fd",str(gate_read), "--database",str(self.store.database),"--artifacts",str(self.store.artifacts), "--run-id",run_id,"--attempt-token",reservation.attempt_token, "--supervisor-token",str(reservation.supervisor_token),"--stdout",paths["stdout_spool"], @@ -229,7 +229,7 @@ def import_durable(self, run_id: str) -> str: identity=inspect_process(attempt["pid"],attempt["pgid"],attempt["process_start_id"]) if identity=="live": return "live" receipt_path=Path(attempt["exit_record"] or "") - if identity=="ambiguous": + if identity=="ambiguous" and not receipt_path.is_file(): self.store.block_recovery(run_id,attempt["attempt_token"],"attempt runner identity is ambiguous") return "ownership_ambiguous" if not receipt_path.is_file(): @@ -247,27 +247,38 @@ def import_durable(self, run_id: str) -> str: return "recovery_required" try: receipt=json.loads(receipt_path.read_text()) - if type(receipt.get("returncode")) is not int or type(receipt.get("cancelled")) is not bool: + if (type(receipt.get("returncode")) is not int + or type(receipt.get("cancelled")) is not bool + or type(receipt.get("timed_out")) is not bool + or (receipt["cancelled"] and receipt["timed_out"])): raise ValueError("invalid receipt") metadata={name:receipt[name] for name in ("stdout","stderr")} - ids=[] + artifacts=[] for stream,column in (("stdout","stdout_spool"),("stderr","stderr_spool")): data=Path(attempt[column]).read_bytes(); meta=metadata[stream] if hashlib.sha256(data).hexdigest()!=meta["captured_sha256"] or len(data)!=meta["captured_bytes"]: raise ValueError("capture hash mismatch") - logical=f"{attempt['id']}.{stream}"; existing=self.store.artifact_named(run_id,logical) - ids.append(existing["id"] if existing else self.store.store_artifact(run_id,logical,data)) - if not attempt.get("output_metadata"): - self.store.record_attempt_output(run_id,attempt["attempt_token"],ids[0],ids[1],metadata) + logical=f"{attempt['id']}.{stream}" + path,digest,size=self.store.finalize_artifact(run_id,logical,data) + artifacts.append({"name":logical,"path":path,"sha256":digest,"byte_size":size}) receipt_bytes=canonical_json(receipt).encode() - if not self.store.artifact_named(run_id,"result-receipt.json"): - self.store.store_artifact(run_id,"result-receipt.json",receipt_bytes) - terminal="cancelled" if receipt["cancelled"] else ("succeeded" if receipt["returncode"]==0 else "failed") - self.store.finish_attempt(run_id,attempt["attempt_token"],terminal,{"returncode":receipt["returncode"],"receipt":"result-receipt.json"}) - return terminal + path,digest,size=self.store.finalize_artifact(run_id,"result-receipt.json",receipt_bytes) + artifacts.append({"name":"result-receipt.json","path":path,"sha256":digest,"byte_size":size}) + terminal="cancelled" if receipt["cancelled"] else ("failed" if receipt["timed_out"] or receipt["returncode"]!=0 else "succeeded") + payload={"returncode":receipt["returncode"],"receipt":"result-receipt.json"} + if receipt["timed_out"]: payload["error"]="TIMEOUT" + return self.store.commit_durable_import( + run_id,attempt["attempt_token"],artifacts,metadata,terminal,payload, + ) except (OSError,ValueError,KeyError,TypeError,json.JSONDecodeError): - self.store.block_recovery(run_id,attempt["attempt_token"],"durable receipt or capture is invalid") - return "ownership_ambiguous" + try: + self.store.block_recovery(run_id,attempt["attempt_token"],"durable receipt or capture is invalid") + return "ownership_ambiguous" + except ConflictError: + current=self.store.attempt(run_id); run=self.store.run(run_id) + if current and current["attempt_token"]==attempt["attempt_token"] and current["status"]=="finished" and run["state"] in {"succeeded","failed","cancelled"}: + return run["state"] + raise def _terminate_durable(self, handle: DurableAttempt) -> None: attempt=self.store.active_attempt(handle.reservation.run_id) diff --git a/test/core/test_m2_supervisor_gate.py b/test/core/test_m2_supervisor_gate.py index 1a4c49e..3ea895a 100644 --- a/test/core/test_m2_supervisor_gate.py +++ b/test/core/test_m2_supervisor_gate.py @@ -114,6 +114,33 @@ def test_database_fences_second_writer_for_same_worktree(self): self.supervisor.cancel(first_run, handle) self.assertEqual(self.store.run(first_run)["state"], "cancelled") + def test_durable_timeout_is_failure_when_term_handler_exits_zero(self): + run_id, version = self._ready_run("timeout-zero") + code = ( + "import signal,sys,time;" + "signal.signal(signal.SIGTERM,lambda *_:sys.exit(0));" + "time.sleep(30)" + ) + spec = LaunchSpec( + 1, "fake", "cli_exec", (sys.executable, "-c", code), str(self.repo), None, 1, + ExecutionIdentity("fake", "1", "fixture", "fixture", "fixture", "low"), + ) + source = str(ROOT / "plugin" / "core" / "src") + with mock.patch.dict(os.environ, {"PYTHONPATH": source}): + handle = self.supervisor.launch_durable(run_id, version, spec, "owner", "package") + self.assertEqual(self.supervisor.wait_durable(handle, 1), 124) + self.assertEqual(self.store.run(run_id)["state"], "failed") + receipt_artifact = self.store.artifact_named(run_id, "result-receipt.json") + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertTrue(receipt["timed_out"]) + self.assertEqual(receipt["returncode"], 0) + self.assertEqual(receipt["error"], "TIMEOUT") + terminal = self.store.connection.execute( + "SELECT payload FROM events WHERE run_id=? AND type='run.failed'", (run_id,), + ).fetchone() + self.assertEqual(json.loads(terminal["payload"])["error"], "TIMEOUT") + self.assertEqual(self.store.attempt(run_id)["status"], "finished") + def test_recovery_never_signals_an_ambiguous_identity(self): run_id, version = self._ready_run("ambiguous") handle = self.supervisor.launch( diff --git a/test/core/test_service.py b/test/core/test_service.py index db79677..221d3f3 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -1,4 +1,5 @@ import json +import multiprocessing import os from pathlib import Path import signal @@ -14,6 +15,20 @@ from devsquad.service import Service from devsquad.store import ConflictError, Store +from devsquad.supervisor import Supervisor, inspect_process + + +def concurrent_receipt_import(database, artifacts, run_id, barrier, results): + store = None + try: + store = Store(Path(database), Path(artifacts)) + barrier.wait(timeout=10) + results.put(Supervisor(store).import_durable(run_id)) + except Exception as exc: + results.put(f"{type(exc).__name__}: {exc}") + finally: + if store is not None: + store.close() class ServiceTest(unittest.TestCase): @@ -113,6 +128,62 @@ def test_coordinator_crash_imports_runner_receipt_once(self): try: self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM attempts WHERE run_id=?",(started["run_id"],)).fetchone()[0],1) finally: store.close() + def test_two_processes_import_one_runner_receipt_atomically(self): + started=self.service.start(self.task,"receipt-race",_internal_fake_delay=.3) + self.wait_state(started["run_id"],{"running"}) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + attempt=store.attempt(started["run_id"]) + owner=store.connection.execute("SELECT owner_id FROM supervisor_claims WHERE run_id=?",(started["run_id"],)).fetchone()[0] + os.kill(int(owner.split(":",1)[1]),signal.SIGKILL) + receipt=Path(attempt["exit_record"]) + deadline=time.monotonic()+5 + while not receipt.is_file() and time.monotonic() Date: Wed, 9 Sep 2026 21:58:03 +0530 Subject: [PATCH 034/197] test: verify M2 CLI and installed migrations --- plugin/core/src/devsquad/cli.py | 83 +++++++-- test/core/test_cli.py | 287 ++++++++++++++++++++++++++++++++ 2 files changed, 356 insertions(+), 14 deletions(-) create mode 100644 test/core/test_cli.py diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 7fcdd0a..2f13328 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -1,4 +1,4 @@ -"""Small M1 command surface: version, doctor, prepare and classify.""" +"""Versioned JSON command surface for local DevSquad operations.""" from __future__ import annotations @@ -6,16 +6,27 @@ import json import os import sys +import time from pathlib import Path +from typing import Any from . import __version__ from .adapters import AdapterManifest, classify_cli, harness_version, prepare_cli, prepare_native_codex_from_catalog from .contracts import ContractError, envelope, error_payload from .service import Service -from .store import ConflictError +from .store import ConflictError, SchemaVersionError SOURCE_ROOT = Path(__file__).resolve().parents[2] CORE_ROOT = SOURCE_ROOT if (SOURCE_ROOT / "adapters").is_dir() else Path(sys.prefix) / "share" / "devsquad" +WAIT_POLL_SECONDS = 0.25 +WAIT_EXIT_CODES = { + "succeeded": 0, + "blocked": 2, + "awaiting_host": 2, + "failed": 3, + "cancelled": 4, +} +WAIT_ACTIVE_STATES = {"queued", "running", "cancelling"} def manifests() -> list[tuple[Path, AdapterManifest]]: @@ -33,7 +44,14 @@ def command_doctor(_: argparse.Namespace) -> tuple[dict, int]: return envelope(data={"core_version": __version__, "ready": ready, "adapters": rows}), 0 if ready else 1 -def command_prepare(args: argparse.Namespace) -> dict: +def _read_json(path: str, label: str) -> Any: + try: + return json.loads(Path(path).read_text()) + except (OSError, UnicodeError, json.JSONDecodeError) as exc: + raise ContractError(f"cannot read {label}: {exc}") from exc + + +def command_prepare(args: argparse.Namespace) -> tuple[dict, int]: manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") transport = args.transport or manifest.transport if transport == "native_protocol": @@ -41,7 +59,7 @@ def command_prepare(args: argparse.Namespace) -> dict: raise ContractError("native preparation requires Codex, --catalog-file, --model and --effort") binary = manifest.resolve_binary() version = harness_version(binary) if binary else None - snapshot = json.loads(Path(args.catalog_file).read_text()) + snapshot = _read_json(args.catalog_file, "catalog file") spec = prepare_native_codex_from_catalog(manifest, snapshot, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout, harness_version_value=version or "unknown") elif transport == "cli_exec": spec = prepare_cli(manifest, prompt=args.prompt, cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) @@ -50,10 +68,15 @@ def command_prepare(args: argparse.Namespace) -> dict: return envelope(data=spec.to_dict()), 0 -def command_classify(args: argparse.Namespace) -> dict: +def command_classify(args: argparse.Namespace) -> tuple[dict, int]: manifest = AdapterManifest.load(CORE_ROOT / "adapters" / args.adapter / "adapter.json") spec = prepare_cli(manifest, prompt="classification", cwd=args.cwd, model=args.model, effort=args.effort, permission=args.permission, timeout_seconds=args.timeout) - result = classify_cli(spec, returncode=args.returncode, stdout=Path(args.stdout_file).read_text(), stderr=Path(args.stderr_file).read_text()) + try: + stdout = Path(args.stdout_file).read_text() + stderr = Path(args.stderr_file).read_text() + except (OSError, UnicodeError) as exc: + raise ContractError(f"cannot read classification output: {exc}") from exc + result = classify_cli(spec, returncode=args.returncode, stdout=stdout, stderr=stderr) return envelope(data=result.to_dict()), 0 @@ -62,8 +85,35 @@ def _service(args: argparse.Namespace) -> Service: def command_start(args: argparse.Namespace) -> tuple[dict, int]: - task = json.loads(Path(args.task_file).read_text()) - return envelope(data=_service(args).start(task, args.idempotency_key, args.supersedes_run)), 0 + task = _read_json(args.task_file, "task file") + service = _service(args) + started = service.start(task, args.idempotency_key, args.supersedes_run) + if not args.wait: + return envelope(data=started), 0 + run_id = started["run_id"] + try: + while True: + status = service.status(run_id) + state = status.get("state") + if state in WAIT_EXIT_CODES: + return envelope(data=status), WAIT_EXIT_CODES[state] + if state not in WAIT_ACTIVE_STATES: + raise RuntimeError(f"service returned unsupported run state: {state!r}") + time.sleep(WAIT_POLL_SECONDS) + except KeyboardInterrupt: + cancel_command = f"squad cancel {run_id}" + print( + f"Stopped observing run {run_id}; the run was not cancelled and remains saved. " + f"To cancel it explicitly, run: {cancel_command}", + file=sys.stderr, + ) + return envelope(data={ + "run_id": run_id, + "state": started.get("state"), + "observation_stopped": True, + "cancelled": False, + "next_action": cancel_command, + }), 130 def command_status(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).status(args.run)), 0 @@ -71,7 +121,7 @@ def command_events(args: argparse.Namespace) -> tuple[dict, int]: return envelop def command_result(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).result(args.run)), 0 def command_cancel(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).cancel(args.run)), 0 def command_resume(args: argparse.Namespace) -> tuple[dict, int]: - recovery = json.loads(Path(args.recovery_file).read_text()) if args.recovery_file else None + recovery = _read_json(args.recovery_file, "recovery file") if args.recovery_file else None return envelope(data=_service(args).resume(args.run, recovery)), 0 @@ -103,7 +153,7 @@ def parser() -> argparse.ArgumentParser: cmd.add_argument("--stderr-file", required=True) cmd.set_defaults(func=fn) runtime_default = os.environ.get("DEVSQUAD_RUNTIME_DIR", str(Path.home() / ".devsquad" / "runtime")) - start = sub.add_parser("start"); start.add_argument("--task-file", required=True); start.add_argument("--idempotency-key", required=True); start.add_argument("--supersedes-run"); start.add_argument("--json", action="store_true"); start.add_argument("--runtime-dir", default=runtime_default); start.set_defaults(func=command_start) + start = sub.add_parser("start"); start.add_argument("--task-file", required=True); start.add_argument("--idempotency-key", required=True); start.add_argument("--supersedes-run"); start.add_argument("--wait", action="store_true"); start.add_argument("--json", action="store_true"); start.add_argument("--runtime-dir", default=runtime_default); start.set_defaults(func=command_start) for name, fn in (("status",command_status),("result",command_result),("cancel",command_cancel),("resume",command_resume)): cmd=sub.add_parser(name); cmd.add_argument("run"); cmd.add_argument("--json",action="store_true"); cmd.add_argument("--runtime-dir",default=runtime_default) if name == "resume": cmd.add_argument("--recovery-file") @@ -118,10 +168,15 @@ def main(argv: list[str] | None = None) -> int: response, code = args.func(args) print(json.dumps(response, sort_keys=True)) return code - except (ContractError, OSError, json.JSONDecodeError) as exc: - code = getattr(exc, "code", "INPUT_INVALID") - print(json.dumps(envelope(error=error_payload(code, str(exc))), sort_keys=True)) - return 75 if isinstance(exc, ConflictError) else 64 + except (ConflictError, SchemaVersionError) as exc: + print(json.dumps(envelope(error=error_payload(exc.code, str(exc))), sort_keys=True)) + return 75 + except ContractError as exc: + print(json.dumps(envelope(error=error_payload(exc.code, str(exc))), sort_keys=True)) + return 64 + except Exception as exc: + print(json.dumps(envelope(error=error_payload("INTERNAL_ERROR", str(exc))), sort_keys=True)) + return 1 if __name__ == "__main__": diff --git a/test/core/test_cli.py b/test/core/test_cli.py new file mode 100644 index 0000000..08d8c7e --- /dev/null +++ b/test/core/test_cli.py @@ -0,0 +1,287 @@ +import contextlib +import io +import json +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad import cli +from devsquad.contracts import ContractError +from devsquad.store import ConflictError, SchemaVersionError + + +class CliTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-cli-") + self.root = Path(self.temp.name) + self.runtime = self.root / "runtime" + self.task_file = self.root / "task.json" + self.task_file.write_text('{"schema_version": 1}\n') + + def tearDown(self): + self.temp.cleanup() + + def invoke(self, argv, service=None): + stdout, stderr = io.StringIO(), io.StringIO() + patcher = mock.patch.object(cli, "Service", return_value=service) if service is not None else contextlib.nullcontext() + with patcher, contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + code = cli.main(argv) + lines = stdout.getvalue().splitlines() + self.assertEqual(len(lines), 1, stdout.getvalue()) + return code, json.loads(lines[0]), stderr.getvalue() + + def assert_success_envelope(self, payload, data): + self.assertEqual(payload, { + "schema_version": 1, + "ok": True, + "data": data, + "error": None, + }) + + def assert_error_envelope(self, payload, code, message): + self.assertEqual(payload, { + "schema_version": 1, + "ok": False, + "data": None, + "error": { + "code": code, + "message": message, + "retryable": False, + "details": {}, + }, + }) + + def test_start_returns_exact_envelope_and_forwards_inputs(self): + service = mock.Mock() + response = {"run_id": "run-1", "state": "queued", "created": True} + service.start.return_value = response + code, payload, stderr = self.invoke([ + "start", "--task-file", str(self.task_file), + "--idempotency-key", "key-1", "--supersedes-run", "old-run", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual(code, 0) + self.assertEqual(stderr, "") + self.assert_success_envelope(payload, response) + service.start.assert_called_once_with({"schema_version": 1}, "key-1", "old-run") + service.status.assert_not_called() + + def test_status_events_result_cancel_and_resume_operations(self): + cases = [ + (["status", "run-1"], "status", ("run-1",), {"run_id": "run-1", "state": "failed"}), + (["events", "run-1", "--after", "7", "--limit", "9"], "events", ("run-1", 7, 9), {"events": [], "next_cursor": 7}), + (["result", "run-1"], "result", ("run-1",), {"run_id": "run-1", "ready": False}), + (["cancel", "run-1"], "cancel", ("run-1",), {"run_id": "run-1", "state": "cancelling"}), + ] + for arguments, method_name, expected_args, response in cases: + with self.subTest(command=arguments[0]): + service = mock.Mock() + getattr(service, method_name).return_value = response + code, payload, stderr = self.invoke(arguments + ["--runtime-dir", str(self.runtime), "--json"], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + getattr(service, method_name).assert_called_once_with(*expected_args) + + recovery_file = self.root / "recovery.json" + recovery_file.write_text('{"attempt_id":"a-1","disposition":"confirm_dead"}\n') + service = mock.Mock() + response = {"run_id": "run-1", "disposition": "continued", "launched": True} + service.resume.return_value = response + code, payload, stderr = self.invoke([ + "resume", "run-1", "--recovery-file", str(recovery_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.resume.assert_called_once_with("run-1", {"attempt_id": "a-1", "disposition": "confirm_dead"}) + + def test_parser_and_json_file_failures_are_input_errors(self): + code, payload, _ = self.invoke(["status"]) + self.assertEqual(code, 64) + self.assertEqual(payload["error"]["code"], "INPUT_INVALID") + + malformed = self.root / "malformed.json" + malformed.write_text("{") + code, payload, _ = self.invoke([ + "start", "--task-file", str(malformed), "--idempotency-key", "key", + "--runtime-dir", str(self.runtime), "--json", + ]) + self.assertEqual(code, 64) + self.assertEqual(payload["error"]["code"], "INPUT_INVALID") + self.assertIn("cannot read task file", payload["error"]["message"]) + + def test_contract_conflict_schema_and_runtime_errors_have_exact_exits(self): + cases = [ + (ContractError("bad input"), 64, "INPUT_INVALID"), + (ConflictError("run version changed"), 75, "CONFLICT"), + (SchemaVersionError("database is newer"), 75, "SCHEMA_UNSUPPORTED"), + (OSError("disk failed"), 1, "INTERNAL_ERROR"), + (RuntimeError("unexpected"), 1, "INTERNAL_ERROR"), + ] + for exception, expected_exit, expected_code in cases: + with self.subTest(exception=type(exception).__name__): + service = mock.Mock() + service.status.side_effect = exception + code, payload, stderr = self.invoke([ + "status", "run-1", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (expected_exit, "")) + self.assert_error_envelope(payload, expected_code, str(exception)) + + def test_wait_maps_every_terminal_and_paused_state_to_contract_exit(self): + expected = { + "succeeded": 0, + "blocked": 2, + "awaiting_host": 2, + "failed": 3, + "cancelled": 4, + } + for state, expected_exit in expected.items(): + with self.subTest(state=state): + service = mock.Mock() + service.start.return_value = {"run_id": "run-1", "state": "queued", "created": True} + status = {"run_id": "run-1", "state": state, "version": 3} + service.status.return_value = status + code, payload, stderr = self.invoke([ + "start", "--task-file", str(self.task_file), "--idempotency-key", "key", + "--wait", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (expected_exit, "")) + self.assert_success_envelope(payload, status) + service.cancel.assert_not_called() + + def test_wait_observes_active_states_until_terminal(self): + service = mock.Mock() + service.start.return_value = {"run_id": "run-1", "state": "queued", "created": True} + service.status.side_effect = [ + {"run_id": "run-1", "state": "queued"}, + {"run_id": "run-1", "state": "running"}, + {"run_id": "run-1", "state": "succeeded"}, + ] + with mock.patch.object(cli.time, "sleep") as sleep: + code, payload, _ = self.invoke([ + "start", "--task-file", str(self.task_file), "--idempotency-key", "key", + "--wait", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual(code, 0) + self.assertEqual(payload["data"]["state"], "succeeded") + self.assertEqual(service.status.call_count, 3) + self.assertEqual(sleep.call_count, 2) + service.cancel.assert_not_called() + + def test_wait_keyboard_interrupt_stops_observation_without_cancelling(self): + service = mock.Mock() + service.start.return_value = {"run_id": "run-42", "state": "running", "created": True} + service.status.side_effect = KeyboardInterrupt() + code, payload, stderr = self.invoke([ + "start", "--task-file", str(self.task_file), "--idempotency-key", "key", + "--wait", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual(code, 130) + self.assert_success_envelope(payload, { + "run_id": "run-42", + "state": "running", + "observation_stopped": True, + "cancelled": False, + "next_action": "squad cancel run-42", + }) + self.assertIn("run-42", stderr) + self.assertIn("squad cancel run-42", stderr) + self.assertIn("not cancelled", stderr) + service.cancel.assert_not_called() + + def test_wait_rejects_unknown_service_state_as_internal_error(self): + service = mock.Mock() + service.start.return_value = {"run_id": "run-1", "state": "queued", "created": True} + service.status.return_value = {"run_id": "run-1", "state": "mystery"} + code, payload, _ = self.invoke([ + "start", "--task-file", str(self.task_file), "--idempotency-key", "key", + "--wait", "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual(code, 1) + self.assertEqual(payload["error"]["code"], "INTERNAL_ERROR") + + +class InstalledWheelMigrationTest(unittest.TestCase): + @staticmethod + def build_python(): + candidates = [ + os.environ.get("DEVSQUAD_BUILD_PYTHON"), + sys.executable, + str(Path.home() / ".cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3"), + shutil.which("python3.13"), + shutil.which("python3.12"), + shutil.which("python3.11"), + ] + for candidate in dict.fromkeys(value for value in candidates if value): + result = subprocess.run([ + candidate, "-c", + "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68", + ], text=True, capture_output=True) + if result.returncode == 0: + return candidate + return None + + def test_installed_wheel_contains_and_applies_migration_four(self): + build_python = self.build_python() + if build_python is None: + self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") + with tempfile.TemporaryDirectory(prefix="devsquad-wheel-") as directory: + root = Path(directory) + source = root / "core" + shutil.copytree(ROOT / "plugin/core", source) + wheels = root / "wheels" + wheels.mkdir() + subprocess.run([ + build_python, "-m", "pip", "wheel", str(source), + "--wheel-dir", str(wheels), "--no-index", "--no-deps", "--no-build-isolation", + ], check=True, text=True, capture_output=True) + wheel = next(wheels.glob("devsquad_core-*.whl")) + environment = os.environ.copy() + environment.pop("PYTHONPATH", None) + venv = root / "venv" + subprocess.run([build_python, "-m", "venv", str(venv)], check=True, env=environment) + python = venv / ("Scripts/python.exe" if os.name == "nt" else "bin/python") + subprocess.run([ + str(python), "-m", "pip", "install", "--no-index", "--no-deps", str(wheel), + ], check=True, text=True, capture_output=True, env=environment) + probe = r''' +from importlib.resources import files +from pathlib import Path +import sqlite3 +import sys +from devsquad.store import Store + +root = Path(sys.argv[1]) +root.mkdir(parents=True) +database = root / "state.sqlite3" +connection = sqlite3.connect(database) +migrations = files("devsquad.migrations") +for version, name in ((1, "001_initial.sql"), (2, "002_supervisor.sql"), (3, "003_durable_io.sql")): + connection.executescript(migrations.joinpath(name).read_text()) + connection.execute("INSERT INTO schema_migrations(version, applied_at) VALUES(?, 'fixture')", (version,)) +connection.commit() +connection.close() +store = Store(database, root / "artifacts") +try: + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 4 + columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} + assert {"package_path", "package_digest", "supersedes_run_id"} <= columns +finally: + store.close() +''' + subprocess.run([ + str(python), "-P", "-c", probe, str(root / "probe-runtime"), + ], check=True, text=True, capture_output=True, cwd=root, env=environment) + + +if __name__ == "__main__": + unittest.main() From 846458083ac87723177fa06cc1b7fc87689df817 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 9 Sep 2026 22:00:10 +0530 Subject: [PATCH 035/197] docs: checkpoint M2 adversarial hardening --- docs/plans/engineering-team/M2-STATUS.md | 32 +++++++++++++----------- docs/plans/engineering-team/RESUME.md | 27 +++++++++++++------- docs/plans/engineering-team/backlog.json | 9 +++++++ 3 files changed, 45 insertions(+), 23 deletions(-) diff --git a/docs/plans/engineering-team/M2-STATUS.md b/docs/plans/engineering-team/M2-STATUS.md index 55e7bf4..392a9ae 100644 --- a/docs/plans/engineering-team/M2-STATUS.md +++ b/docs/plans/engineering-team/M2-STATUS.md @@ -12,24 +12,28 @@ service checkpoint is `a0794a9`. | Preparing-owner fencing and cancellation | Runs remain `queued` with a private `preparing` phase; stale completion after cancel conflicts; generic events cannot bypass the fence | verified offline | | Transactional projections and events | Compare-and-swap run version and append-only event commit together under concurrent writers | verified offline | | Atomic hash-verified artifacts | Content-addressed files finalize before reference; references increment run version with an event; duplicates and terminal mutation cannot clobber prior content | verified offline | -| Schema migration and future refusal | A real version-1 fixture upgrades to version 2; newer unsupported versions are rejected | verified offline | +| Schema migration and future refusal | Source fixtures upgrade from versions 1 and 3; an installed wheel applies migration 004 from a schema-3 fixture; newer unsupported versions are rejected | verified offline | | Supervisor and writer fencing | Transactional claims allow one supervisor and one active writer per worktree; ambiguous ownership retains the database fence | verified offline | | Strong process identity and recovery | Darwin start second+microsecond identity is stable; live children remain owned without relaunch; dead and reused identities receive distinct recovery dispositions and reused IDs are never signalled | verified offline | | Bounded process lifecycle | Direct argv runs in a new session; heartbeat, PID/PGID/start identity, token and package digest persist; stdout/stderr drain continuously with truncation and full-stream hashes | verified offline | -| Timeout and cancellation | Intent precedes verified TERM/KILL; TERM-resistant root and descendant disappear before completion; repeated terminal cancel is harmless | verified offline | +| Timeout and cancellation | Intent precedes verified TERM/KILL; TERM-resistant root and descendant disappear before completion; durable timeout remains failed when TERM exits 0 | partial: pre-attempt receipts and cross-process cancel cleanup remain | | Durable gated launch | A persisted attempt runner and an inner worker gate prevent task execution before strong runner and child identity records; the runner owns timeout, cancel polling, bounded spool files and an fsynced exit receipt | verified offline | -| Coordinator-loss recovery | A separate coordinator is killed while the runner lives; resume does not relaunch it, and a completed receipt is imported once with one attempt | verified offline | -| Frozen package | Every regular runtime asset is hashed, copied atomically, fsynced, stored on the run and verified before initial launch or resume | verified offline | +| Coordinator-loss recovery | A separate coordinator is killed while the runner lives; two processes import one completed receipt atomically without relaunch or duplicate events | partial: preparing/launching and orphan-child recovery remain | +| Frozen package | Every regular runtime asset is hashed, copied atomically, fsynced, stored on the run and verified before initial launch or resume; `-P` regressions prevent repository shadowing of internal modules | verified offline | | Public workflow guard | Public branch-review tasks end with `CAPABILITY_UNAVAILABLE` until M3; only the private test argument can invoke the M2 fake step | verified offline | -| Result and event reads | Status is a transactional run/attempt snapshot; cursors always report the last consumed position and `has_more`; terminal service results verify referenced blob hashes and require a durable receipt for attempted runs | verified offline | -| Predecessor link | A superseded run must be terminal and belong to the same canonical Git project; the link is committed with preparation | verified offline | +| Result and event reads | Status is a transactional run/attempt snapshot; cursors always report the last consumed position and `has_more`; attempted-run receipts and hashes are checked | partial: failed-preflight/pre-attempt cancellation receipts remain | +| Predecessor link | Fixture execution validates terminal same-project predecessors and stores the link | partial: public capability-failure path still bypasses validation | +| CLI envelope and wait behavior | Exact v1 envelopes/exit codes, M2 operation dispatch, observation-only Ctrl-C, and terminal `start --wait` mappings | verified offline | -The recovery checkpoint `a0794a9` passed 67 core tests and 202 shell -assertions. The current service candidate passes 69 core tests, including a -coordinator-crash receipt-import case. Final shell and wheel evidence will be -recorded at the acceptance checkpoint. +Checkpoints `5aa2e74` and `f0d29a7` pass 82 core tests and 202 shell +assertions. The core suite includes an installed-wheel schema-3-to-4 migration, +a two-process receipt-import race, repository-shadow resistance, and a timeout +whose TERM handler exits zero. Several service tests still emit detached +subprocess `ResourceWarning`s; M2 is not accepted while those ownership and +recovery gaps remain. -Remaining acceptance work is the independent service adversarial gate, -cross-process race expansion, CLI envelope verification and installed-wheel -migration 3 check. This is not a claim of a working engineering workflow; -branch-review execution begins in M3. +Remaining acceptance work is crashed preparation/launch reconciliation, +receipts for every terminal path, predecessor/stdin/cancel invariants, +cross-process start/writer/crash races, host-handoff storage and claim fencing, +ResourceWarning cleanup, and a post-fix independent Astra gate. This is not a +claim of a working engineering workflow; branch-review execution begins in M3. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 8372ecf..c4cb279 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -26,18 +26,25 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne - Cleanup-race hardening checkpoint: `df955f4`. Cleanup inventory fails closed, timeout/cancel identity races retain ownership fencing, and 62 core tests pass. - Durable service checkpoint `cefd193` is explicit WIP and gated-runner - checkpoint `a0794a9` passed 67 core tests plus 202 shell assertions. The current local + checkpoint `a0794a9` passed 67 core tests plus 202 shell assertions. The + preserved recovery candidate is `f76df41`. Adversarial hardening checkpoint + `5aa2e74` and CLI/wheel checkpoint `f0d29a7` pass 82 core tests plus all 202 + shell assertions. The current implementation checkpoint replaces the unsafe anonymous-pipe launch with a persisted, gated attempt runner and migration 003 durable paths. The runner cannot launch the internal fixture worker until its strong identity and spool paths are committed; it owns output drains, heartbeat, cancel polling and an atomic exit record. Ordered migration discovery, transactional status, stable event cursors and fenced preparation failure are implemented. - Recovery/import of a runner exit record, predecessor validation, package - pin verification, and the result-receipt barrier are now implemented and - covered by 69 core tests. The independent adversarial service gate, - cross-process race expansion, CLI envelopes and wheel migration-3 check - remain open. Public branch-review execution is explicitly failed as + Recovery/import of a runner exit record is transactionally idempotent under + two-process races; timeouts cannot become success after a clean TERM exit; + repository-local Python packages cannot shadow the frozen internal runner. + Exact CLI envelopes, `start --wait`, and an installed-wheel schema-3-to-4 + migration gate are covered. Remaining M2 work includes crashed preparation + and launch recovery, pre-attempt terminal receipts, complete predecessor + validation, durable stdin/cancel cleanup races, cross-process start/writer + coverage, host-handoff claim fencing, the detached-process ResourceWarnings, + and a post-fix Astra gate. Public branch-review execution is explicitly failed as `CAPABILITY_UNAVAILABLE` until M3; only the internal test hook can run the fake step. M2 remains in progress. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. @@ -76,9 +83,11 @@ The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed e 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. 2. Read the M2 section of [IMPLEMENTATION.md](IMPLEMENTATION.md), [M2-STATUS.md](M2-STATUS.md), and the corresponding contracts before - editing. Continue by implementing durable receipt import/recovery without - relaunch, then package/predecessor/result invariants and the cross-process - crash tests. Do not redo the accepted store or supervisor foundations. + editing. Continue with crashed `preparing`/`launching` reconciliation and + durable receipts for every terminal path, then predecessor/cancel/stdin + invariants, the cross-process crash matrix, and host-handoff claim fencing. + Do not redo the accepted store or supervisor foundations or the fixes in + `5aa2e74`/`f0d29a7`. 3. Preserve M1 limitations and process-ownership boundaries. M1 acceptance is not a claim that the full product exists, and no additional native probe is needed for the accepted gate. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index b1c36ba..364bc69 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -47,6 +47,15 @@ "artifact": "M2-STATUS.md", "recorded_at": "2026-09-08T00:00:00+05:30", "availability": "tracked tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "f0d29a7", + "command_or_action": "82 core tests, 202 shell assertions, installed-wheel schema-3-to-4 migration, two-process receipt import, repository-shadow and timeout adversarial regressions", + "outcome": "Durable import and CLI/wheel slices pass; M2 remains in progress for crash-phase recovery, all-terminal receipts, cancellation/stdin races, host claims and independent re-review", + "artifact": "M2-STATUS.md", + "recorded_at": "2026-09-09T21:58:27+05:30", + "availability": "tracked tests" } ], "blocker": null From 144b0b22bb196d9bb6ec87b46a5806b86a5c3e8e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 15 Sep 2026 10:53:53 +0530 Subject: [PATCH 036/197] fix: recover interrupted preflight with receipts --- plugin/core/src/devsquad/service.py | 101 +++++++++++++++--- plugin/core/src/devsquad/store.py | 160 ++++++++++++++++++++++++++-- test/core/test_m2_gate_review.py | 3 +- test/core/test_service.py | 74 ++++++++++++- test/core/test_store.py | 56 +++++++++- 5 files changed, 365 insertions(+), 29 deletions(-) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 1e2d5ec..fa493bf 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -92,26 +92,77 @@ def oid(ref: str) -> str: snapshot["internal_fake_delay"] = internal_delay return snapshot + def _continue_preparation( + self, + store: Store, + run_id: str, + fencing_token: int, + submitted: dict[str, Any], + ) -> tuple[tuple[int, Path, str] | None, dict[str, Any] | None]: + snapshot = None + supersedes_run_id = submitted.get("supersedes_run_id") + try: + task = submitted["task"] + internal_delay = submitted.get("_internal_fake_delay") + validate_task(task, require_existing_repo=True) + snapshot = self._resolve_snapshot(task, internal_delay) + if internal_delay is None: + error = { + "error": "CAPABILITY_UNAVAILABLE", + "message": "branch-review workflow is introduced in M3", + } + store.fail_preparation( + run_id, + fencing_token, + error, + mutable_snapshot=snapshot, + supersedes_run_id=supersedes_run_id, + ) + return None, error + package, digest = self._freeze_package() + version = store.complete_preparation( + run_id, + fencing_token, + snapshot, + package_path=str(package), + package_digest=digest, + supersedes_run_id=supersedes_run_id, + ) + return (version, package, digest), None + except Exception as exc: + error = {"error": "PREPARATION_FAILED", "message": str(exc)} + try: + # A rejected predecessor remains in submitted_request but is not + # published as a valid supersession relation. + store.fail_preparation( + run_id, + fencing_token, + error, + mutable_snapshot=snapshot, + ) + except ConflictError: + # Cancellation or another recovery owner may have fenced us. + raise exc + return None, error + def start(self, task: dict[str, Any], idempotency_key: str, supersedes_run_id: str | None = None, *, _internal_fake_delay: float | None = None) -> dict[str, Any]: validate_task(task, require_existing_repo=True) submitted = {"task": task, "supersedes_run_id": supersedes_run_id} + if _internal_fake_delay is not None: + submitted["_internal_fake_delay"] = _internal_fake_delay store = self._store() try: claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}") if not claim.created: return {"run_id": claim.run_id, "state": store.run(claim.run_id)["state"], "created": False} - try: - snapshot = self._resolve_snapshot(task, _internal_fake_delay) - if _internal_fake_delay is None: - store.fail_preparation(claim.run_id, claim.fencing_token or 0, {"error": "CAPABILITY_UNAVAILABLE", "message": "branch-review workflow is introduced in M3"}) - return {"run_id": claim.run_id, "state": "failed", "created": True} - package, digest = self._freeze_package() - version = store.complete_preparation(claim.run_id, claim.fencing_token or 0, snapshot, package_path=str(package), package_digest=digest, supersedes_run_id=supersedes_run_id) - except Exception as exc: - store.fail_preparation(claim.run_id, claim.fencing_token or 0, {"error": "PREPARATION_FAILED", "message": str(exc)}) - raise + launch, error = self._continue_preparation( + store, claim.run_id, claim.fencing_token or 0, submitted, + ) finally: store.close() + if launch is None: + return {"run_id": claim.run_id, "state": "failed", "created": True, "error": error} + version, package, digest = launch self._spawn_daemon(claim.run_id, version, package, digest) return {"run_id": claim.run_id, "state": "queued", "created": True} @@ -142,9 +193,8 @@ def result(self, run_id: str) -> dict[str, Any]: run, artifacts = store.result_snapshot(run_id) if run["state"] not in TERMINAL_STATES: return {"run_id":run_id,"ready":False,"state":run["state"],"artifacts":[]} - if run["state"] != "failed" or run.get("package_digest"): - if not any(item["name"]=="result-receipt.json" for item in artifacts): - raise ConflictError("terminal result has no durable receipt") + if not any(item["name"]=="result-receipt.json" for item in artifacts): + raise ConflictError("terminal result has no durable receipt") for item in artifacts: path=Path(item["path"]) if not path.is_file() or hashlib.sha256(path.read_bytes()).hexdigest()!=item["sha256"]: @@ -167,6 +217,8 @@ def cancel(self, run_id: str) -> dict[str, Any]: def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[str, Any]: store = self._store() + launch: tuple[int, Path, str] | None = None + preparation_error: dict[str, Any] | None = None try: run = store.run(run_id) if run["state"] in TERMINAL_STATES: raise ConflictError("terminal run cannot resume; start a superseding run") @@ -175,7 +227,15 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st attempt=store.attempt(run_id) disposition = Supervisor(store).import_durable(run_id) if attempt and attempt.get("exit_record") else Supervisor(store).recover(run_id) return {"run_id": run_id, "disposition": disposition, "launched": False} - if run["state"] == "queued" and run["phase"] is None: + if run["state"] == "queued" and run["phase"] == "preparing": + submitted = json.loads(run["submitted_request"]) + claim = store.reclaim_preparation( + run_id, run["version"], f"preflight-recovery:{os.getpid()}", + ) + launch, preparation_error = self._continue_preparation( + store, run_id, claim.fencing_token, submitted, + ) + elif run["state"] == "queued" and run["phase"] is None: version = run["version"] elif run["state"] == "blocked": if not isinstance(recovery,dict) or set(recovery)!={"attempt_id","disposition"} or recovery["disposition"] not in {"confirm_dead","retain_ownership"}: @@ -190,5 +250,16 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st else: raise ConflictError("run is not resumable") finally: store.close() - package,digest=self._verified_package(run); self._spawn_daemon(run_id, version, package, digest) + if run["phase"] == "preparing": + if launch is None: + return { + "run_id": run_id, + "disposition": "preparation_failed", + "launched": False, + "error": preparation_error, + } + version,package,digest=launch + else: + package,digest=self._verified_package(run) + self._spawn_daemon(run_id, version, package, digest) return {"run_id": run_id, "disposition": "continued", "launched": True} diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 30be3e1..7278415 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -48,6 +48,13 @@ class AttemptReservation: version: int +@dataclass(frozen=True) +class PreparationClaim: + run_id: str + fencing_token: int + version: int + + def _utc_now() -> str: return datetime.now(timezone.utc).isoformat() @@ -163,6 +170,89 @@ def claim_start(self, worktree: Path, idempotency_key: str, submitted_request: A self.connection.execute("ROLLBACK") raise + def reclaim_preparation(self, run_id: str, expected_version: int, owner_id: str) -> PreparationClaim: + """Take over an interrupted preflight while fencing its former owner.""" + if type(expected_version) is not int or expected_version < 1 or not owner_id: + raise ContractError("preparation recovery requires a version and owner") + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,c.fencing_token " + "FROM runs r JOIN claims c ON c.run_id=r.id WHERE r.id=?", + (run_id,), + ).fetchone() + if (not row or row["state"] != "queued" or row["phase"] != "preparing" + or row["version"] != expected_version): + raise ConflictError("preparation is no longer recoverable at that version") + token, version, now = row["fencing_token"] + 1, expected_version + 1, _utc_now() + self.connection.execute( + "UPDATE claims SET fencing_token=?,owner_id=?,active=1,claimed_at=? WHERE run_id=?", + (token, owner_id, now, run_id), + ) + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({"owner_id": owner_id, "fencing_token": token}) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.preparation_reclaimed',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return PreparationClaim(run_id, token, version) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def _terminal_receipt(self, run_id: str, terminal_state: str, phase: str, payload: Any) -> tuple[Path, str, int, str]: + finished_at = _utc_now() + receipt = { + "schema_version": 1, + "run_id": run_id, + "state": terminal_state, + "phase": phase, + "attempt_id": None, + "returncode": None, + "cancelled": terminal_state == "cancelled", + "timed_out": False, + "error": payload if terminal_state == "failed" else None, + "finished_at": finished_at, + } + path, digest, size = self.finalize_artifact( + run_id, "result-receipt.json", canonical_json(receipt).encode(), + ) + return path, digest, size, finished_at + + def _reference_terminal_receipt( + self, + run_id: str, + version: int, + path: Path, + digest: str, + size: int, + now: str, + ) -> int: + artifact_id = str(uuid.uuid4()) + self.connection.execute( + "INSERT INTO artifacts(id,run_id,name,path,sha256,byte_size,created_at) " + "VALUES(?,?,'result-receipt.json',?,?,?,?)", + (artifact_id, run_id, str(path), digest, size, now), + ) + version += 1 + artifact_event = canonical_json({ + "artifact_id": artifact_id, + "name": "result-receipt.json", + "sha256": digest, + "byte_size": size, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'artifact.recorded',?,?)", + (run_id, version, artifact_event, now), + ) + return version + def complete_preparation(self, run_id: str, fencing_token: int, mutable_snapshot: Any, *, package_path: str | None = None, package_digest: str | None = None, supersedes_run_id: str | None = None) -> int: snapshot = canonical_json(mutable_snapshot) self.connection.execute("BEGIN IMMEDIATE") @@ -188,20 +278,51 @@ def complete_preparation(self, run_id: str, fencing_token: int, mutable_snapshot self.connection.execute("ROLLBACK") raise - def fail_preparation(self, run_id: str, fencing_token: int, error: Any) -> int: + def fail_preparation( + self, + run_id: str, + fencing_token: int, + error: Any, + *, + mutable_snapshot: Any | None = None, + supersedes_run_id: str | None = None, + ) -> int: encoded = canonical_json(error) self.connection.execute("BEGIN IMMEDIATE") try: row = self.connection.execute( - "SELECT r.state,r.phase,r.version,c.fencing_token,c.active FROM runs r JOIN claims c ON c.run_id=r.id WHERE r.id=?", + "SELECT r.project_id,r.state,r.phase,r.version,c.fencing_token,c.active " + "FROM runs r JOIN claims c ON c.run_id=r.id WHERE r.id=?", (run_id,), ).fetchone() if not row or row["state"] != "queued" or row["phase"] != "preparing" or not row["active"] or row["fencing_token"] != fencing_token: raise ConflictError("preparation failure is stale or cancelled") - version, now = row["version"] + 1, _utc_now() - self.connection.execute("UPDATE runs SET state='failed',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, run_id)) + if supersedes_run_id is not None: + predecessor = self.connection.execute( + "SELECT project_id,state FROM runs WHERE id=?", (supersedes_run_id,), + ).fetchone() + if (not predecessor or predecessor["project_id"] != row["project_id"] + or predecessor["state"] not in TERMINAL_STATES): + raise ConflictError("superseded run must be terminal and belong to the same project") + path, digest, size, now = self._terminal_receipt(run_id, "failed", "preparing", error) + version = self._reference_terminal_receipt( + run_id, row["version"], path, digest, size, now, + ) + version += 1 + snapshot = canonical_json(mutable_snapshot) if mutable_snapshot is not None else None + self.connection.execute( + "UPDATE runs SET mutable_snapshot=COALESCE(?,mutable_snapshot),supersedes_run_id=?," + "state='failed',phase=NULL,version=?,updated_at=? WHERE id=?", + (snapshot, supersedes_run_id, version, now, run_id), + ) self.connection.execute("UPDATE claims SET active=0 WHERE run_id=?", (run_id,)) - self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.failed',?,?)", (run_id, version, encoded, now)) + terminal_payload = dict(error) if isinstance(error, dict) else {"error": error} + terminal_payload["receipt"] = "result-receipt.json" + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.failed',?,?)", + (run_id, version, canonical_json(terminal_payload), now), + ) self.connection.execute("COMMIT") return version except Exception: @@ -219,10 +340,17 @@ def cancel_preparing(self, run_id: str) -> int: return row["version"] if row["state"] != "queued" or row["phase"] != "preparing": raise ConflictError("run is no longer preparing") - version, now = row["version"] + 1, _utc_now() + path, digest, size, now = self._terminal_receipt(run_id, "cancelled", "preparing", None) + version = self._reference_terminal_receipt( + run_id, row["version"], path, digest, size, now, + ) + 1 self.connection.execute("UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, run_id)) self.connection.execute("UPDATE claims SET active=0,fencing_token=fencing_token+1 WHERE run_id=?", (run_id,)) - self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelled','{}',?)", (run_id, version, now)) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, canonical_json({"receipt": "result-receipt.json"}), now), + ) self.connection.execute("COMMIT") return version except Exception: @@ -612,9 +740,14 @@ def cancel_queued(self, run_id: str) -> int: self.connection.execute("COMMIT"); return row["version"] if row["state"] != "queued" or row["phase"] is not None: raise ConflictError("queued run is owned by another operation") - version, now = row["version"] + 1, _utc_now() + path,digest,size,now=self._terminal_receipt(run_id,"cancelled","queued",None) + version=self._reference_terminal_receipt(run_id,row["version"],path,digest,size,now)+1 self.connection.execute("UPDATE runs SET state='cancelled',version=?,updated_at=? WHERE id=?", (version, now, run_id)) - self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelled','{}',?)", (run_id, version, now)) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id,version,canonical_json({"receipt":"result-receipt.json"}),now), + ) self.connection.execute("COMMIT"); return version except Exception: self.connection.execute("ROLLBACK"); raise @@ -624,11 +757,16 @@ def cancel_launching(self, run_id: str) -> int: try: row=self.connection.execute("SELECT state,phase,version FROM runs WHERE id=?",(run_id,)).fetchone() if not row or row["state"]!="queued" or row["phase"]!="launching": raise ConflictError("run is not launching") - version,now=row["version"]+1,_utc_now() + path,digest,size,now=self._terminal_receipt(run_id,"cancelled","launching",None) + version=self._reference_terminal_receipt(run_id,row["version"],path,digest,size,now)+1 self.connection.execute("UPDATE attempts SET status='recovery_required',finished_at=? WHERE run_id=? AND status='reserved'",(now,run_id)) self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?",(run_id,)) self.connection.execute("UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?",(version,now,run_id)) - self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelled','{}',?)",(run_id,version,now)) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id,version,canonical_json({"receipt":"result-receipt.json"}),now), + ) self.connection.execute("COMMIT"); return version except Exception: self.connection.execute("ROLLBACK"); raise diff --git a/test/core/test_m2_gate_review.py b/test/core/test_m2_gate_review.py index 1ce5470..ef1f538 100644 --- a/test/core/test_m2_gate_review.py +++ b/test/core/test_m2_gate_review.py @@ -93,7 +93,8 @@ def test_artifact_reference_is_versioned_and_terminal_run_is_immutable(self): count = self.store.connection.execute( "SELECT COUNT(*) FROM artifacts WHERE run_id=?", (self.claim.run_id,), ).fetchone()[0] - self.assertEqual(count, 1) + self.assertEqual(count, 2) + self.assertIsNotNone(self.store.artifact_named(self.claim.run_id, "result-receipt.json")) class StoreInitializationReview(unittest.TestCase): diff --git a/test/core/test_service.py b/test/core/test_service.py index 221d3f3..43e3178 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -67,6 +67,25 @@ def test_start_is_idempotent_and_result_events_are_durable(self): self.assertEqual(len(page["events"]),2); self.assertIsNotNone(page["next_cursor"]) with self.assertRaises(ConflictError): self.service.resume(first["run_id"]) + def test_abandoned_preparation_is_reclaimed_from_the_submitted_request(self): + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + claim=store.claim_start( + self.repo, + "abandoned-preparation", + {"task":self.task,"supersedes_run_id":None,"_internal_fake_delay":.01}, + "dead-preflight-owner", + ) + finally: + store.close() + resumed=self.service.resume(claim.run_id) + self.assertTrue(resumed["launched"]) + self.assertEqual(resumed["disposition"],"continued") + self.wait_state(claim.run_id,{"succeeded"}) + events=self.service.events(claim.run_id)["events"] + self.assertEqual(sum(event["type"]=="run.preparation_reclaimed" for event in events),1) + self.assertTrue(self.service.result(claim.run_id)["ready"]) + def test_new_process_inspects_run_after_launching_process_exits(self): task_file=self.root/"task.json"; task_file.write_text(json.dumps(self.task)) env=os.environ.copy(); env["PYTHONPATH"]=str(ROOT/"plugin/core/src") @@ -109,7 +128,60 @@ def test_orphan_artifact_is_not_a_result(self): store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") try: store.finalize_artifact(started["run_id"],"orphan",b"bytes") finally: store.close() - self.assertEqual(self.service.result(started["run_id"])["artifacts"],[]) + artifacts=self.service.result(started["run_id"])["artifacts"] + self.assertEqual([item["name"] for item in artifacts],["result-receipt.json"]) + + def test_pre_attempt_failure_and_cancellation_have_durable_receipts(self): + failed=self.service.start(self.task,"capability-unavailable") + self.assertEqual(failed["state"],"failed") + failure_result=self.service.result(failed["run_id"]) + self.assertTrue(failure_result["ready"]) + failure_receipt=json.loads(Path(failure_result["artifacts"][0]["path"]).read_text()) + self.assertEqual(failure_receipt["state"],"failed") + self.assertEqual(failure_receipt["error"]["error"],"CAPABILITY_UNAVAILABLE") + + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + preparing=store.claim_start( + self.repo,"cancel-preparing",{"task":self.task,"supersedes_run_id":None},"owner", + ) + finally: + store.close() + cancelled=self.service.cancel(preparing.run_id) + self.assertEqual(cancelled["state"],"cancelled") + cancellation_result=self.service.result(preparing.run_id) + cancellation_receipt=json.loads(Path(cancellation_result["artifacts"][0]["path"]).read_text()) + self.assertTrue(cancellation_receipt["cancelled"]) + self.assertEqual(cancellation_receipt["phase"],"preparing") + + def test_invalid_predecessor_fails_with_run_context_and_receipt(self): + started=self.service.start( + self.task,"invalid-predecessor","does-not-exist",_internal_fake_delay=.01, + ) + self.assertEqual(started["state"],"failed") + self.assertEqual(started["error"]["error"],"PREPARATION_FAILED") + self.assertIn("superseded run",started["error"]["message"]) + result=self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + self.assertEqual([item["name"] for item in result["artifacts"]],["result-receipt.json"]) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + self.assertIsNone(store.run(started["run_id"])["supersedes_run_id"]) + finally: + store.close() + + def test_post_claim_snapshot_failure_has_run_context_and_receipt(self): + task=json.loads(json.dumps(self.task)) + task["project"]["base_ref"]="refs/heads/does-not-exist" + started=self.service.start(task,"invalid-moving-ref",_internal_fake_delay=.01) + self.assertEqual(started["state"],"failed") + self.assertEqual(started["error"]["error"],"PREPARATION_FAILED") + self.assertIn("does not resolve",started["error"]["message"]) + result=self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + receipt=json.loads(Path(result["artifacts"][0]["path"]).read_text()) + self.assertEqual(receipt["run_id"],started["run_id"]) + self.assertEqual(receipt["error"],started["error"]) def test_coordinator_crash_imports_runner_receipt_once(self): started=self.service.start(self.task,"receipt-recovery",_internal_fake_delay=.3) diff --git a/test/core/test_store.py b/test/core/test_store.py index 86eaa05..5ee1ff3 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -61,7 +61,61 @@ def test_request_is_claimed_before_snapshot_and_cancel_fences_late_preflight(sel with self.assertRaises(ConflictError): self.store.complete_preparation(first.run_id, first.fencing_token, {"branch": "moved"}) run = self.store.run(first.run_id) - self.assertEqual((run["state"], run["version"]), ("cancelled", 2)) + self.assertEqual((run["state"], run["version"]), ("cancelled", 3)) + self.assertIsNotNone(self.store.artifact_named(first.run_id, "result-receipt.json")) + + def test_two_recovery_claimants_yield_one_owner_and_fence_the_old_token(self): + original = self.store.claim_start(self.repo, "recover", {"task": "fixed"}, "old-owner") + expected_version = self.store.run(original.run_id)["version"] + barrier = threading.Barrier(2) + successes, conflicts = [], [] + + def reclaim(owner): + connection = Store(self.database, self.artifacts) + try: + barrier.wait() + successes.append(connection.reclaim_preparation(original.run_id, expected_version, owner)) + except ConflictError as exc: + conflicts.append(exc) + finally: + connection.close() + + threads = [threading.Thread(target=reclaim, args=(owner,)) for owner in ("recovery-a", "recovery-b")] + for thread in threads: thread.start() + for thread in threads: thread.join() + + self.assertEqual(len(successes), 1) + self.assertEqual(len(conflicts), 1) + self.assertEqual(successes[0].fencing_token, (original.fencing_token or 0) + 1) + with self.assertRaises(ConflictError): + self.store.complete_preparation(original.run_id, original.fencing_token, {"branch": "stale"}) + version = self.store.complete_preparation( + original.run_id, successes[0].fencing_token, {"branch": "recovered"}, + ) + self.assertEqual(version, successes[0].version + 1) + + def test_queued_and_launching_cancellation_publish_one_receipt(self): + queued = self.store.claim_start(self.repo, "cancel-queued", {"task": 1}, "owner") + queued_version = self.store.complete_preparation( + queued.run_id, queued.fencing_token, {"branch": "frozen"}, + ) + cancelled_version = self.store.cancel_queued(queued.run_id) + self.assertEqual(cancelled_version, queued_version + 2) + self.assertIsNotNone(self.store.artifact_named(queued.run_id, "result-receipt.json")) + self.assertEqual(self.store.cancel_queued(queued.run_id), cancelled_version) + + launching = self.store.claim_start(self.repo, "cancel-launching", {"task": 2}, "owner") + launching_version = self.store.complete_preparation( + launching.run_id, launching.fencing_token, {"branch": "frozen"}, + ) + reservation = self.store.reserve_attempt( + launching.run_id, launching_version, "supervisor", "package-digest", + ) + final_version = self.store.cancel_launching(launching.run_id) + self.assertEqual(final_version, reservation.version + 2) + self.assertIsNotNone(self.store.artifact_named(launching.run_id, "result-receipt.json")) + with self.assertRaises(ConflictError): + self.store.mark_attempt_running(reservation, 10, 10, "stale-process") def test_event_and_projection_compare_and_swap_share_transaction(self): claim = self.store.claim_start(self.repo, "events", {"task": "x"}, "owner") From c6a7a233a88ecfb67ea0476018529270910150f8 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 15 Sep 2026 10:59:42 +0530 Subject: [PATCH 037/197] fix: bind recovered preflight to claimed project --- plugin/core/src/devsquad/service.py | 23 ++++++-- plugin/core/src/devsquad/store.py | 38 ++++++++++++ test/core/test_service.py | 91 +++++++++++++++++++++++++++++ test/core/test_store.py | 11 ++++ 4 files changed, 157 insertions(+), 6 deletions(-) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index fa493bf..e0a1438 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -74,8 +74,12 @@ def _verified_package(self, run: dict[str, Any]) -> tuple[Path,str]: return package,run["package_digest"] @staticmethod - def _resolve_snapshot(task: dict[str, Any], internal_delay: float | None) -> dict[str, Any]: - repo = Path(task["project"]["repo_path"]).resolve(strict=True) + def _resolve_snapshot( + task: dict[str, Any], + internal_delay: float | None, + resolved_repo: Path | None = None, + ) -> dict[str, Any]: + repo = resolved_repo or Path(task["project"]["repo_path"]).resolve(strict=True) def oid(ref: str) -> str: result = subprocess.run(["git", "-C", str(repo), "rev-parse", "--verify", f"{ref}^{{commit}}"], text=True, capture_output=True, check=False) if result.returncode != 0: raise ContractError(f"Git ref does not resolve to a commit: {ref}") @@ -101,11 +105,17 @@ def _continue_preparation( ) -> tuple[tuple[int, Path, str] | None, dict[str, Any] | None]: snapshot = None supersedes_run_id = submitted.get("supersedes_run_id") + validated_supersedes_run_id = None try: task = submitted["task"] internal_delay = submitted.get("_internal_fake_delay") validate_task(task, require_existing_repo=True) - snapshot = self._resolve_snapshot(task, internal_delay) + store.validate_predecessor(run_id, fencing_token, supersedes_run_id) + validated_supersedes_run_id = supersedes_run_id + worktree = store.preparation_worktree( + run_id, fencing_token, Path(task["project"]["repo_path"]), + ) + snapshot = self._resolve_snapshot(task, internal_delay, worktree) if internal_delay is None: error = { "error": "CAPABILITY_UNAVAILABLE", @@ -116,7 +126,7 @@ def _continue_preparation( fencing_token, error, mutable_snapshot=snapshot, - supersedes_run_id=supersedes_run_id, + supersedes_run_id=validated_supersedes_run_id, ) return None, error package, digest = self._freeze_package() @@ -132,13 +142,14 @@ def _continue_preparation( except Exception as exc: error = {"error": "PREPARATION_FAILED", "message": str(exc)} try: - # A rejected predecessor remains in submitted_request but is not - # published as a valid supersession relation. + # Preserve validated lineage through unrelated failures; a + # rejected predecessor remains only in submitted_request. store.fail_preparation( run_id, fencing_token, error, mutable_snapshot=snapshot, + supersedes_run_id=validated_supersedes_run_id, ) except ConflictError: # Cancellation or another recovery owner may have fenced us. diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 7278415..627590f 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -205,6 +205,44 @@ def reclaim_preparation(self, run_id: str, expected_version: int, owner_id: str) self.connection.execute("ROLLBACK") raise + def preparation_worktree(self, run_id: str, fencing_token: int, submitted_path: Path) -> Path: + """Resolve a submitted path only if it still names the claimed project.""" + row = self.connection.execute( + "SELECT r.state,r.phase,r.worktree_path,proj.git_common_dir,c.fencing_token,c.active " + "FROM runs r JOIN projects proj ON proj.id=r.project_id " + "JOIN claims c ON c.run_id=r.id WHERE r.id=?", + (run_id,), + ).fetchone() + if (not row or row["state"] != "queued" or row["phase"] != "preparing" + or not row["active"] or row["fencing_token"] != fencing_token): + raise ConflictError("preparation claim is stale or cancelled") + try: + resolved = submitted_path.resolve(strict=True) + except OSError as exc: + raise ContractError("claimed project worktree is no longer available") from exc + if str(resolved) != row["worktree_path"]: + raise ContractError("project repo_path no longer resolves to the claimed worktree") + if str(git_common_dir(resolved)) != row["git_common_dir"]: + raise ContractError("claimed worktree no longer belongs to the persisted project") + return resolved + + def validate_predecessor(self, run_id: str, fencing_token: int, supersedes_run_id: str | None) -> None: + if supersedes_run_id is None: + return + row = self.connection.execute( + "SELECT r.project_id,r.state,r.phase,c.fencing_token,c.active," + "predecessor.project_id AS predecessor_project,predecessor.state AS predecessor_state " + "FROM runs r JOIN claims c ON c.run_id=r.id " + "LEFT JOIN runs predecessor ON predecessor.id=? WHERE r.id=?", + (supersedes_run_id, run_id), + ).fetchone() + if (not row or row["state"] != "queued" or row["phase"] != "preparing" + or not row["active"] or row["fencing_token"] != fencing_token): + raise ConflictError("preparation claim is stale or cancelled") + if (row["predecessor_project"] != row["project_id"] + or row["predecessor_state"] not in TERMINAL_STATES): + raise ConflictError("superseded run must be terminal and belong to the same project") + def _terminal_receipt(self, run_id: str, terminal_state: str, phase: str, payload: Any) -> tuple[Path, str, int, str]: finished_at = _utc_now() receipt = { diff --git a/test/core/test_service.py b/test/core/test_service.py index 43e3178..528ea7c 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -86,6 +86,46 @@ def test_abandoned_preparation_is_reclaimed_from_the_submitted_request(self): self.assertEqual(sum(event["type"]=="run.preparation_reclaimed" for event in events),1) self.assertTrue(self.service.result(claim.run_id)["ready"]) + def test_preparation_recovery_rejects_a_retargeted_repository_symlink(self): + link=self.root/"repo-link" + link.symlink_to(self.repo,target_is_directory=True) + task=json.loads(json.dumps(self.task)) + task["project"]["repo_path"]=str(link) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + claim=store.claim_start( + link, + "retargeted-repository", + {"task":task,"supersedes_run_id":None,"_internal_fake_delay":.01}, + "dead-preflight-owner", + ) + finally: + store.close() + + other=self.root/"other-repo" + subprocess.run(["git","init","-q",str(other)],check=True) + subprocess.run(["git","-C",str(other),"config","user.email","test@example.invalid"],check=True) + subprocess.run(["git","-C",str(other),"config","user.name","Test"],check=True) + (other/"profiles.json").write_text('{"source":"other"}\n') + (other/"policy.json").write_text('{"source":"other"}\n') + subprocess.run(["git","-C",str(other),"add","."],check=True) + subprocess.run(["git","-C",str(other),"commit","-qm","other"],check=True) + link.unlink() + link.symlink_to(other,target_is_directory=True) + + resumed=self.service.resume(claim.run_id) + self.assertFalse(resumed["launched"]) + self.assertEqual(resumed["disposition"],"preparation_failed") + self.assertIn("claimed worktree",resumed["error"]["message"]) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + run=store.run(claim.run_id) + self.assertEqual(run["state"],"failed") + self.assertEqual(run["worktree_path"],str(self.repo.resolve())) + self.assertIsNone(run["mutable_snapshot"]) + finally: + store.close() + def test_new_process_inspects_run_after_launching_process_exits(self): task_file=self.root/"task.json"; task_file.write_text(json.dumps(self.task)) env=os.environ.copy(); env["PYTHONPATH"]=str(ROOT/"plugin/core/src") @@ -170,6 +210,42 @@ def test_invalid_predecessor_fails_with_run_context_and_receipt(self): finally: store.close() + def test_public_active_and_foreign_predecessor_matrix(self): + terminal=self.service.start(self.task,"terminal-predecessor") + public=self.service.start(self.task,"public-successor",terminal["run_id"]) + + with mock.patch.object(self.service,"_spawn_daemon",return_value=0): + active=self.service.start(self.task,"active-predecessor",_internal_fake_delay=.2) + against_active=self.service.start( + self.task,"against-active",active["run_id"],_internal_fake_delay=.01, + ) + + other=self.root/"foreign-repo" + subprocess.run(["git","init","-q",str(other)],check=True) + subprocess.run(["git","-C",str(other),"config","user.email","test@example.invalid"],check=True) + subprocess.run(["git","-C",str(other),"config","user.name","Test"],check=True) + (other/"profiles.json").write_text("{}\n") + (other/"policy.json").write_text("{}\n") + subprocess.run(["git","-C",str(other),"add","."],check=True) + subprocess.run(["git","-C",str(other),"commit","-qm","foreign"],check=True) + foreign_task=json.loads(json.dumps(self.task)) + foreign_task["project"]["repo_path"]=str(other) + foreign=self.service.start(foreign_task,"foreign-predecessor") + against_foreign=self.service.start( + self.task,"against-foreign",foreign["run_id"],_internal_fake_delay=.01, + ) + + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + self.assertEqual(store.run(public["run_id"])["supersedes_run_id"],terminal["run_id"]) + self.assertIsNone(store.run(against_active["run_id"])["supersedes_run_id"]) + self.assertIsNone(store.run(against_foreign["run_id"])["supersedes_run_id"]) + finally: + store.close() + self.assertEqual(against_active["state"],"failed") + self.assertEqual(against_foreign["state"],"failed") + self.service.cancel(active["run_id"]) + def test_post_claim_snapshot_failure_has_run_context_and_receipt(self): task=json.loads(json.dumps(self.task)) task["project"]["base_ref"]="refs/heads/does-not-exist" @@ -183,6 +259,21 @@ def test_post_claim_snapshot_failure_has_run_context_and_receipt(self): self.assertEqual(receipt["run_id"],started["run_id"]) self.assertEqual(receipt["error"],started["error"]) + def test_valid_predecessor_survives_an_unrelated_snapshot_failure(self): + predecessor=self.service.start(self.task,"lineage-predecessor") + task=json.loads(json.dumps(self.task)) + task["project"]["target_ref"]="refs/heads/does-not-exist" + replacement=self.service.start( + task,"lineage-replacement",predecessor["run_id"],_internal_fake_delay=.01, + ) + self.assertEqual(replacement["state"],"failed") + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + run=store.run(replacement["run_id"]) + self.assertEqual(run["supersedes_run_id"],predecessor["run_id"]) + finally: + store.close() + def test_coordinator_crash_imports_runner_receipt_once(self): started=self.service.start(self.task,"receipt-recovery",_internal_fake_delay=.3) self.wait_state(started["run_id"],{"running"}) diff --git a/test/core/test_store.py b/test/core/test_store.py index 5ee1ff3..bf45960 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -1,3 +1,4 @@ +import hashlib import json from pathlib import Path import sqlite3 @@ -117,6 +118,16 @@ def test_queued_and_launching_cancellation_publish_one_receipt(self): with self.assertRaises(ConflictError): self.store.mark_attempt_running(reservation, 10, 10, "stale-process") + for run_id,phase in ((queued.run_id,"queued"),(launching.run_id,"launching")): + artifact=self.store.artifact_named(run_id,"result-receipt.json") + content=Path(artifact["path"]).read_bytes() + receipt=json.loads(content) + self.assertEqual(hashlib.sha256(content).hexdigest(),artifact["sha256"]) + self.assertEqual( + (receipt["run_id"],receipt["state"],receipt["phase"],receipt["cancelled"]), + (run_id,"cancelled",phase,True), + ) + def test_event_and_projection_compare_and_swap_share_transaction(self): claim = self.store.claim_start(self.repo, "events", {"task": "x"}, "owner") with self.assertRaises(ConflictError): From f9f2ffa0f76c17c7374d523347db09f0c02a6d38 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 15 Sep 2026 11:05:46 +0530 Subject: [PATCH 038/197] fix: recover gated launches and durable stdin --- plugin/core/src/devsquad/attempt_runner.py | 12 +++- plugin/core/src/devsquad/service.py | 10 ++- plugin/core/src/devsquad/store.py | 45 +++++++++++++ plugin/core/src/devsquad/supervisor.py | 38 +++++++++-- test/core/test_m2_supervisor_gate.py | 29 ++++++++- test/core/test_service.py | 73 ++++++++++++++++++++++ 6 files changed, 197 insertions(+), 10 deletions(-) diff --git a/plugin/core/src/devsquad/attempt_runner.py b/plugin/core/src/devsquad/attempt_runner.py index e98ab02..cb7b127 100644 --- a/plugin/core/src/devsquad/attempt_runner.py +++ b/plugin/core/src/devsquad/attempt_runner.py @@ -25,13 +25,19 @@ def _drain(stream, capture: Path, limit: int, result: dict): result.update(total_bytes=total,captured_bytes=kept,truncated=total>kept,full_sha256=digest.hexdigest(),captured_sha256=hashlib.sha256(capture.read_bytes()).hexdigest()) def main(argv=None): - p=argparse.ArgumentParser(); p.add_argument("--gate-fd",type=int,required=True); p.add_argument("--database",type=Path,required=True); p.add_argument("--artifacts",type=Path,required=True); p.add_argument("--run-id",required=True); p.add_argument("--attempt-token",required=True); p.add_argument("--supervisor-token",type=int,required=True); p.add_argument("--stdout",type=Path,required=True); p.add_argument("--stderr",type=Path,required=True); p.add_argument("--exit-record",type=Path,required=True); p.add_argument("--child-record",type=Path,required=True); p.add_argument("--limit",type=int,required=True); p.add_argument("--timeout",type=float,required=True); p.add_argument("--grace",type=float,required=True); p.add_argument("command",nargs=argparse.REMAINDER) + p=argparse.ArgumentParser(); p.add_argument("--gate-fd",type=int,required=True); p.add_argument("--stdin-fd",type=int); p.add_argument("--database",type=Path,required=True); p.add_argument("--artifacts",type=Path,required=True); p.add_argument("--run-id",required=True); p.add_argument("--attempt-token",required=True); p.add_argument("--supervisor-token",type=int,required=True); p.add_argument("--stdout",type=Path,required=True); p.add_argument("--stderr",type=Path,required=True); p.add_argument("--exit-record",type=Path,required=True); p.add_argument("--child-record",type=Path,required=True); p.add_argument("--limit",type=int,required=True); p.add_argument("--timeout",type=float,required=True); p.add_argument("--grace",type=float,required=True); p.add_argument("command",nargs=argparse.REMAINDER) a=p.parse_args(argv); command=a.command[1:] if a.command[:1]==["--"] else a.command with os.fdopen(a.gate_fd,"rb",closefd=True) as gate: - if gate.read(1)!=b"1": return 125 + if gate.read(1)!=b"1": + if a.stdin_fd is not None: os.close(a.stdin_fd) + return 125 child_gate_read,child_gate_write=os.pipe() gated=[os.sys.executable,"-P","-m","devsquad.worker_gate","--gate-fd",str(child_gate_read),"--",*command] - child=subprocess.Popen(gated,stdout=subprocess.PIPE,stderr=subprocess.PIPE,start_new_session=True,pass_fds=(child_gate_read,)) + stdin_source=subprocess.DEVNULL if a.stdin_fd is None else a.stdin_fd + try: + child=subprocess.Popen(gated,stdin=stdin_source,stdout=subprocess.PIPE,stderr=subprocess.PIPE,start_new_session=True,pass_fds=(child_gate_read,)) + finally: + if a.stdin_fd is not None: os.close(a.stdin_fd) os.close(child_gate_read) started=process_start_identity(child.pid) if started is None: diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index e0a1438..81dd21a 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -9,6 +9,7 @@ import subprocess import sys import tempfile +import threading from typing import Any from .contracts import ContractError @@ -109,9 +110,9 @@ def _continue_preparation( try: task = submitted["task"] internal_delay = submitted.get("_internal_fake_delay") - validate_task(task, require_existing_repo=True) store.validate_predecessor(run_id, fencing_token, supersedes_run_id) validated_supersedes_run_id = supersedes_run_id + validate_task(task, require_existing_repo=True) worktree = store.preparation_worktree( run_id, fencing_token, Path(task["project"]["repo_path"]), ) @@ -183,6 +184,11 @@ def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, diges log_dir=self.runtime/"private-logs"; log_dir.mkdir(parents=True,exist_ok=True) with (log_dir/f"{run_id}.supervisor.log").open("ab",buffering=0) as diagnostic: process = subprocess.Popen(command, cwd=self.runtime, env=environment, stdin=subprocess.DEVNULL, stdout=diagnostic, stderr=diagnostic, start_new_session=True, close_fds=True) + threading.Thread( + target=process.wait, + name=f"devsquad-reap-{run_id}", + daemon=True, + ).start() return process.pid def status(self, run_id: str) -> dict[str, Any]: @@ -246,6 +252,8 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st launch, preparation_error = self._continue_preparation( store, run_id, claim.fencing_token, submitted, ) + elif run["state"] == "queued" and run["phase"] == "launching": + version = store.recover_launching(run_id, run["version"]) elif run["state"] == "queued" and run["phase"] is None: version = run["version"] elif run["state"] == "blocked": diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 627590f..d449416 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -764,6 +764,51 @@ def fail_launch(self, reservation: AttemptReservation, reason: str) -> int: self.connection.execute("ROLLBACK") raise + def recover_launching(self, run_id: str, expected_version: int) -> int: + """Fence an abandoned pre-gate reservation and make the run launchable.""" + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status FROM attempts WHERE run_id=? ORDER BY created_at DESC LIMIT 1", + (run_id,), + ).fetchone() + claim = self.connection.execute( + "SELECT active FROM supervisor_claims WHERE run_id=?", (run_id,), + ).fetchone() + if (not run or run["state"] != "queued" or run["phase"] != "launching" + or run["version"] != expected_version or not attempt + or attempt["status"] != "reserved" or not claim or not claim["active"]): + raise ConflictError("launch reservation is no longer recoverable") + version, now = expected_version + 1, _utc_now() + self.connection.execute( + "UPDATE attempts SET status='recovery_required',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET phase=NULL,version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "reason": "abandoned gated launch reservation", + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.launch_recovered',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + def active_attempt(self, run_id: str) -> dict[str, Any] | None: row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status IN ('reserved','running','cancelling','ownership_ambiguous') ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() return dict(row) if row else None diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 64bfbe8..32ca8f0 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -7,6 +7,7 @@ import hashlib import os import signal +import stat import subprocess import sys import threading @@ -19,6 +20,26 @@ from .store import AttemptReservation, ConflictError, Store, canonical_json +def _open_stdin_artifact(path: str) -> BinaryIO: + candidate = Path(path) + if candidate.is_symlink(): + raise ContractError("stdin artifact must be a regular non-symlink file") + flags = os.O_RDONLY + if hasattr(os, "O_NOFOLLOW"): + flags |= os.O_NOFOLLOW + try: + descriptor = os.open(candidate, flags) + except OSError as exc: + raise ContractError("stdin artifact must be a readable regular non-symlink file") from exc + try: + if not stat.S_ISREG(os.fstat(descriptor).st_mode): + raise ContractError("stdin artifact must be a regular non-symlink file") + return os.fdopen(descriptor, "rb") + except Exception: + os.close(descriptor) + raise + + def process_start_identity(pid: int) -> str | None: if sys.platform.startswith("linux"): try: @@ -133,10 +154,7 @@ def launch(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: try: stdin_stream: Any = subprocess.DEVNULL if spec.stdin_path is not None: - stdin_path = os.path.realpath(spec.stdin_path) - if os.path.islink(spec.stdin_path) or not os.path.isfile(stdin_path): - raise ContractError("stdin artifact must be a regular non-symlink file") - stdin_stream = open(stdin_path, "rb") + stdin_stream = _open_stdin_artifact(spec.stdin_path) try: process = subprocess.Popen(list(spec.argv), cwd=spec.cwd, env=environment, stdin=stdin_stream, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True) finally: @@ -182,7 +200,10 @@ def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, o "stdout_spool":"stdout.capture", "stderr_spool":"stderr.capture", "stdout_meta":"stdout.meta.json", "stderr_meta":"stderr.meta.json", "exit_record":"exit.json", "child_record":"child.json"}.items()} gate_read, gate_write = os.pipe() process = None + stdin_stream = None try: + if spec.stdin_path is not None: + stdin_stream = _open_stdin_artifact(spec.stdin_path) command=[sys.executable,"-P","-m","devsquad.attempt_runner","--gate-fd",str(gate_read), "--database",str(self.store.database),"--artifacts",str(self.store.artifacts), "--run-id",run_id,"--attempt-token",reservation.attempt_token, @@ -190,8 +211,13 @@ def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, o "--stderr",paths["stderr_spool"],"--exit-record",paths["exit_record"], "--child-record",paths["child_record"],"--limit",str(self.output_limit), "--timeout",str(spec.timeout_seconds),"--grace",str(self.grace_seconds),"--",*spec.argv] + if stdin_stream is not None: + command[4:4] = ["--stdin-fd", str(stdin_stream.fileno())] environment=os.environ.copy(); environment.update(spec.environment) - process=subprocess.Popen(command,cwd=spec.cwd,env=environment,stdin=subprocess.DEVNULL,stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL,pass_fds=(gate_read,),start_new_session=True) + passed_fds=(gate_read,) if stdin_stream is None else (gate_read,stdin_stream.fileno()) + process=subprocess.Popen(command,cwd=spec.cwd,env=environment,stdin=subprocess.DEVNULL,stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL,pass_fds=passed_fds,start_new_session=True) + if stdin_stream is not None: + stdin_stream.close(); stdin_stream=None os.close(gate_read) started=process_start_identity(process.pid) if started is None: raise RuntimeError("gated child has no strong process identity") @@ -199,6 +225,8 @@ def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, o os.write(gate_write,b"1"); os.close(gate_write) return DurableAttempt(reservation,process,paths) except Exception: + if stdin_stream is not None: + stdin_stream.close() try: os.close(gate_write) except OSError: pass for fd in (gate_read,): diff --git a/test/core/test_m2_supervisor_gate.py b/test/core/test_m2_supervisor_gate.py index 3ea895a..ec1bfec 100644 --- a/test/core/test_m2_supervisor_gate.py +++ b/test/core/test_m2_supervisor_gate.py @@ -17,7 +17,7 @@ ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) -from devsquad.contracts import ExecutionIdentity, LaunchSpec +from devsquad.contracts import ContractError, ExecutionIdentity, LaunchSpec from devsquad.store import ConflictError, Store from devsquad.supervisor import Supervisor, inspect_process, process_start_identity @@ -141,6 +141,33 @@ def test_durable_timeout_is_failure_when_term_handler_exits_zero(self): self.assertEqual(json.loads(terminal["payload"])["error"], "TIMEOUT") self.assertEqual(self.store.attempt(run_id)["status"], "finished") + def test_durable_launch_passes_an_opened_regular_stdin_artifact(self): + run_id,version=self._ready_run("durable-stdin") + stdin_path=self.root/"request input.json" + payload=b'{"request":"exact bytes"}\n' + stdin_path.write_bytes(payload) + code="import sys;sys.stdout.buffer.write(sys.stdin.buffer.read())" + spec=self._spec(sys.executable,"-c",code,stdin_path=str(stdin_path)) + source=str(ROOT/"plugin"/"core"/"src") + with mock.patch.dict(os.environ,{"PYTHONPATH":source}): + handle=self.supervisor.launch_durable(run_id,version,spec,"owner","package") + self.assertEqual(self.supervisor.wait_durable(handle,2),0) + attempt=self.store.attempt(run_id) + artifact=self.store.connection.execute( + "SELECT path FROM artifacts WHERE id=?",(attempt["stdout_artifact_id"],), + ).fetchone() + self.assertEqual(Path(artifact["path"]).read_bytes(),payload) + + linked_run,linked_version=self._ready_run("durable-stdin-symlink") + linked=self.root/"linked-input" + linked.symlink_to(stdin_path) + linked_spec=self._spec(sys.executable,"-c",code,stdin_path=str(linked)) + with self.assertRaises(ContractError): + self.supervisor.launch_durable( + linked_run,linked_version,linked_spec,"owner","package", + ) + self.assertEqual(self.store.run(linked_run)["state"],"blocked") + def test_recovery_never_signals_an_ambiguous_identity(self): run_id, version = self._ready_run("ambiguous") handle = self.supervisor.launch( diff --git a/test/core/test_service.py b/test/core/test_service.py index 528ea7c..347621e 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -31,6 +31,12 @@ def concurrent_receipt_import(database, artifacts, run_id, barrier, results): store.close() +def crash_after_attempt_reservation(database, artifacts, run_id, expected_version, package_digest): + store = Store(Path(database), Path(artifacts)) + store.reserve_attempt(run_id, expected_version, "crashed-supervisor", package_digest) + os._exit(23) + + class ServiceTest(unittest.TestCase): def setUp(self): self.temp = tempfile.TemporaryDirectory(prefix="devsquad-service-") @@ -147,6 +153,46 @@ def test_enqueue_crash_resumes_once_and_cancel_is_prompt(self): response=self.service.cancel(started["run_id"]); self.assertIn(response["state"],{"cancelling","cancelled"}) self.wait_state(started["run_id"],{"cancelled"}) + def test_process_crash_after_attempt_reservation_is_recoverable(self): + with mock.patch.object(self.service,"_spawn_daemon",return_value=0): + started=self.service.start(self.task,"reservation-crash",_internal_fake_delay=.01) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + run=store.run(started["run_id"]) + expected_version=run["version"] + package_digest=run["package_digest"] + finally: + store.close() + + context=multiprocessing.get_context("spawn") + process=context.Process( + target=crash_after_attempt_reservation, + args=( + str(self.runtime/"state.sqlite3"),str(self.runtime/"artifacts"), + started["run_id"],expected_version,package_digest, + ), + ) + process.start(); process.join(timeout=10) + if process.is_alive(): + process.terminate(); process.join(timeout=2) + self.assertEqual(process.exitcode,23) + self.assertEqual(self.service.status(started["run_id"])["phase"],"launching") + + with mock.patch.object(self.service,"_spawn_daemon",return_value=123) as spawn: + resumed=self.service.resume(started["run_id"]) + self.assertTrue(resumed["launched"]) + spawn.assert_called_once() + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + run=store.run(started["run_id"]) + attempt=store.attempt(started["run_id"]) + self.assertEqual((run["state"],run["phase"]),("queued",None)) + self.assertEqual(attempt["status"],"recovery_required") + event_types=[event["type"] for event in store.events_page(started["run_id"])["events"]] + self.assertEqual(event_types.count("run.launch_recovered"),1) + finally: + store.close() + def test_dead_supervisor_with_live_child_never_relaunches(self): started=self.service.start(self.task,"daemon-crash",_internal_fake_delay=10) status=self.wait_state(started["run_id"],{"running"}) @@ -274,6 +320,33 @@ def test_valid_predecessor_survives_an_unrelated_snapshot_failure(self): finally: store.close() + def test_valid_predecessor_survives_repository_loss_during_recovery(self): + predecessor=self.service.start(self.task,"lost-repo-predecessor") + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + replacement=store.claim_start( + self.repo, + "lost-repo-replacement", + { + "task":self.task, + "supersedes_run_id":predecessor["run_id"], + "_internal_fake_delay":.01, + }, + "dead-preflight-owner", + ) + finally: + store.close() + self.repo.rename(self.root/"repo-moved-away") + resumed=self.service.resume(replacement.run_id) + self.assertEqual(resumed["disposition"],"preparation_failed") + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + run=store.run(replacement.run_id) + self.assertEqual(run["supersedes_run_id"],predecessor["run_id"]) + self.assertIsNotNone(store.artifact_named(replacement.run_id,"result-receipt.json")) + finally: + store.close() + def test_coordinator_crash_imports_runner_receipt_once(self): started=self.service.start(self.task,"receipt-recovery",_internal_fake_delay=.3) self.wait_state(started["run_id"],{"running"}) From ea3d8ffa828fe6c6eb9b03cc665902fc766ba2d7 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 15 Sep 2026 11:08:20 +0530 Subject: [PATCH 039/197] docs: checkpoint M2 crash recovery --- docs/plans/engineering-team/M2-STATUS.md | 35 ++++++++++++------------ docs/plans/engineering-team/RESUME.md | 32 ++++++++++++++-------- docs/plans/engineering-team/backlog.json | 9 ++++++ 3 files changed, 47 insertions(+), 29 deletions(-) diff --git a/docs/plans/engineering-team/M2-STATUS.md b/docs/plans/engineering-team/M2-STATUS.md index 392a9ae..9bf21bf 100644 --- a/docs/plans/engineering-team/M2-STATUS.md +++ b/docs/plans/engineering-team/M2-STATUS.md @@ -9,31 +9,32 @@ service checkpoint is `a0794a9`. | SQLite WAL and packaged schema migration | Fresh install opens the packaged migration; four independent processes concurrently initialize one database | verified offline | | Canonical request idempotency before mutable snapshot resolution | Concurrent connections return one run for the same key/body; a changed body conflicts | verified offline | | Canonical project identity | Main and linked Git worktrees resolve to the same absolute common directory and project ID | verified offline | -| Preparing-owner fencing and cancellation | Runs remain `queued` with a private `preparing` phase; stale completion after cancel conflicts; generic events cannot bypass the fence | verified offline | +| Preparing-owner fencing, recovery and cancellation | Runs remain `queued` with a private `preparing` phase; explicit recovery rotates the fencing token and replays the immutable submitted request; competing reclaimers yield one owner; stale completion after cancel/reclaim conflicts | verified offline | | Transactional projections and events | Compare-and-swap run version and append-only event commit together under concurrent writers | verified offline | | Atomic hash-verified artifacts | Content-addressed files finalize before reference; references increment run version with an event; duplicates and terminal mutation cannot clobber prior content | verified offline | | Schema migration and future refusal | Source fixtures upgrade from versions 1 and 3; an installed wheel applies migration 004 from a schema-3 fixture; newer unsupported versions are rejected | verified offline | | Supervisor and writer fencing | Transactional claims allow one supervisor and one active writer per worktree; ambiguous ownership retains the database fence | verified offline | | Strong process identity and recovery | Darwin start second+microsecond identity is stable; live children remain owned without relaunch; dead and reused identities receive distinct recovery dispositions and reused IDs are never signalled | verified offline | | Bounded process lifecycle | Direct argv runs in a new session; heartbeat, PID/PGID/start identity, token and package digest persist; stdout/stderr drain continuously with truncation and full-stream hashes | verified offline | -| Timeout and cancellation | Intent precedes verified TERM/KILL; TERM-resistant root and descendant disappear before completion; durable timeout remains failed when TERM exits 0 | partial: pre-attempt receipts and cross-process cancel cleanup remain | -| Durable gated launch | A persisted attempt runner and an inner worker gate prevent task execution before strong runner and child identity records; the runner owns timeout, cancel polling, bounded spool files and an fsynced exit receipt | verified offline | -| Coordinator-loss recovery | A separate coordinator is killed while the runner lives; two processes import one completed receipt atomically without relaunch or duplicate events | partial: preparing/launching and orphan-child recovery remain | +| Timeout and cancellation | Intent precedes verified TERM/KILL; TERM-resistant root and descendant disappear before completion; durable timeout remains failed when TERM exits 0; every queued/preparing/launching cancellation atomically publishes a receipt | partial: cross-process cancel/cleanup races remain | +| Durable gated launch | A persisted attempt runner and an inner worker gate prevent task execution before strong runner and child identity records; the runner owns timeout, cancel polling, bounded spool files, exact opened stdin and an fsynced exit receipt | verified offline | +| Coordinator-loss recovery | Preparing work is reclaimed under a rotated token; a real process crash after attempt reservation is fenced and requeued; a separate coordinator killed while the runner lives is not relaunched; two processes import one completed receipt atomically | partial: post-identity/pre-gate and orphan-child recovery remain | | Frozen package | Every regular runtime asset is hashed, copied atomically, fsynced, stored on the run and verified before initial launch or resume; `-P` regressions prevent repository shadowing of internal modules | verified offline | | Public workflow guard | Public branch-review tasks end with `CAPABILITY_UNAVAILABLE` until M3; only the private test argument can invoke the M2 fake step | verified offline | -| Result and event reads | Status is a transactional run/attempt snapshot; cursors always report the last consumed position and `has_more`; attempted-run receipts and hashes are checked | partial: failed-preflight/pre-attempt cancellation receipts remain | -| Predecessor link | Fixture execution validates terminal same-project predecessors and stores the link | partial: public capability-failure path still bypasses validation | +| Result and event reads | Status is a transactional run/attempt snapshot; cursors always report the last consumed position and `has_more`; all terminal paths require a hash-checked result receipt | verified offline | +| Predecessor link | Public and fixture paths validate terminal same-project predecessors before mutable filesystem work; valid lineage survives later preparation failure while missing, active and foreign-project predecessors fail with a receipt | verified offline | | CLI envelope and wait behavior | Exact v1 envelopes/exit codes, M2 operation dispatch, observation-only Ctrl-C, and terminal `start --wait` mappings | verified offline | -Checkpoints `5aa2e74` and `f0d29a7` pass 82 core tests and 202 shell -assertions. The core suite includes an installed-wheel schema-3-to-4 migration, -a two-process receipt-import race, repository-shadow resistance, and a timeout -whose TERM handler exits zero. Several service tests still emit detached -subprocess `ResourceWarning`s; M2 is not accepted while those ownership and -recovery gaps remain. +Checkpoint `f9f2ffa` passes 94 core tests and 202 shell assertions. The core +suite includes an installed-wheel schema-3-to-4 migration, two-process receipt +import and preparation-claim races, real crashes after reservation, +repository-retarget/shadow resistance, exact durable stdin, and a timeout whose +TERM handler exits zero. Detached supervisors are reaped without the earlier +`ResourceWarning`s. An independent review found project-identity and lineage +defects in the first preparation-recovery patch; both have regressions and were +fixed before this checkpoint. -Remaining acceptance work is crashed preparation/launch reconciliation, -receipts for every terminal path, predecessor/stdin/cancel invariants, -cross-process start/writer/crash races, host-handoff storage and claim fencing, -ResourceWarning cleanup, and a post-fix independent Astra gate. This is not a -claim of a working engineering workflow; branch-review execution begins in M3. +Remaining acceptance work is post-identity/pre-gate and orphan-child +reconciliation, cross-process cancel/start/writer races, host-handoff storage +and claim fencing, and a final independent M2 gate. This is not a claim of a +working engineering workflow; branch-review execution begins in M3. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index c4cb279..bfcae23 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,7 +2,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 8, 2026 +## Current position — September 15, 2026 - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` is the published runtime baseline. @@ -29,7 +29,9 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne checkpoint `a0794a9` passed 67 core tests plus 202 shell assertions. The preserved recovery candidate is `f76df41`. Adversarial hardening checkpoint `5aa2e74` and CLI/wheel checkpoint `f0d29a7` pass 82 core tests plus all 202 - shell assertions. The current implementation + shell assertions. Preparation/receipt checkpoint `144b0b2`, reviewed + project-identity fix `c6a7a23`, and gated-launch/stdin checkpoint `f9f2ffa` + now pass 94 core tests plus all 202 shell assertions. The current implementation checkpoint replaces the unsafe anonymous-pipe launch with a persisted, gated attempt runner and migration 003 durable paths. The runner cannot launch the internal fixture worker until its strong identity and spool paths @@ -40,13 +42,19 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne two-process races; timeouts cannot become success after a clean TERM exit; repository-local Python packages cannot shadow the frozen internal runner. Exact CLI envelopes, `start --wait`, and an installed-wheel schema-3-to-4 - migration gate are covered. Remaining M2 work includes crashed preparation - and launch recovery, pre-attempt terminal receipts, complete predecessor - validation, durable stdin/cancel cleanup races, cross-process start/writer - coverage, host-handoff claim fencing, the detached-process ResourceWarnings, - and a post-fix Astra gate. Public branch-review execution is explicitly failed as - `CAPABILITY_UNAVAILABLE` until M3; only the internal test hook can run the - fake step. M2 remains in progress. + migration gate are covered. Interrupted preparation rotates its fencing token + and replays the canonical submitted request without permitting repository + retargeting. Every public pre-attempt terminal path publishes a durable + receipt; public/fixture predecessor validation covers valid, missing, active + and foreign-project links, including unrelated failure and repository-loss + cases. A real crash after gated-launch reservation is recoverable, durable + stdin preserves exact bytes, and detached supervisor handles are reaped. + Independent review found and drove the project-identity/lineage regressions. + Remaining M2 work includes post-identity/pre-gate and orphan-child recovery, + cross-process cancellation/start/writer races, host-handoff claim fencing, + and a final post-fix M2 gate. Public branch-review execution is explicitly + failed as `CAPABILITY_UNAVAILABLE` until M3; only the internal test hook can + run the fake step. M2 remains in progress. - GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. - User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 is in progress. @@ -83,9 +91,9 @@ The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed e 1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. 2. Read the M2 section of [IMPLEMENTATION.md](IMPLEMENTATION.md), [M2-STATUS.md](M2-STATUS.md), and the corresponding contracts before - editing. Continue with crashed `preparing`/`launching` reconciliation and - durable receipts for every terminal path, then predecessor/cancel/stdin - invariants, the cross-process crash matrix, and host-handoff claim fencing. + editing. Continue with migration 005 host-handoff storage and claim fencing, + then post-identity/pre-gate and orphan-child recovery plus the remaining + cross-process cancellation/start/writer crash matrix. Do not redo the accepted store or supervisor foundations or the fixes in `5aa2e74`/`f0d29a7`. 3. Preserve M1 limitations and process-ownership boundaries. M1 acceptance is diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 364bc69..398719c 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -56,6 +56,15 @@ "artifact": "M2-STATUS.md", "recorded_at": "2026-09-09T21:58:27+05:30", "availability": "tracked tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "f9f2ffa", + "command_or_action": "94 core tests with warnings enabled, 202 shell assertions, independent preparation-recovery review, real reservation-process crash, durable stdin and project-retarget regressions", + "outcome": "Preparing and pre-gate launching recovery, pre-attempt receipts, predecessor validation, exact durable stdin and detached-process reaping pass; M2 remains in progress for host claims and remaining process races", + "artifact": "M2-STATUS.md", + "recorded_at": "2026-09-15T11:06:00+05:30", + "availability": "tracked tests" } ], "blocker": null From 1cb7ad63baf3b3fe60115fa015cdc48fea669a3e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 15 Sep 2026 11:10:47 +0530 Subject: [PATCH 040/197] test: prove M2 cross-process fences --- test/core/test_m2_cross_process.py | 155 +++++++++++++++++++++++++++++ 1 file changed, 155 insertions(+) create mode 100644 test/core/test_m2_cross_process.py diff --git a/test/core/test_m2_cross_process.py b/test/core/test_m2_cross_process.py new file mode 100644 index 0000000..950b8c6 --- /dev/null +++ b/test/core/test_m2_cross_process.py @@ -0,0 +1,155 @@ +"""Independent-process races for the public M2 service and writer fences.""" +from __future__ import annotations + +import json +import multiprocessing +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) + +from devsquad.service import Service +from devsquad.store import Store + + +def service_start(runtime, task, key, barrier, results): + try: + barrier.wait(timeout=10) + value = Service(Path(runtime)).start(task, key) + results.put(("ok", value["run_id"], value["created"], value["state"])) + except Exception as exc: + results.put(("error", type(exc).__name__, str(exc))) + + +def service_cancel(runtime, run_id, barrier, results): + try: + barrier.wait(timeout=10) + value = Service(Path(runtime)).cancel(run_id) + results.put(("ok", value["state"], value["version"])) + except Exception as exc: + results.put(("error", type(exc).__name__, str(exc))) + + +def reserve_writer(database, artifacts, run_id, version, owner, barrier, results): + store = None + try: + store = Store(Path(database), Path(artifacts)) + barrier.wait(timeout=10) + reservation = store.reserve_attempt(run_id, version, owner, "package") + results.put(("ok", run_id, reservation.attempt_id)) + except Exception as exc: + results.put(("error", run_id, type(exc).__name__, str(exc))) + finally: + if store is not None: + store.close() + + +class CrossProcessServiceTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-process-races-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], check=True) + subprocess.run(["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True) + (self.repo / "profiles.json").write_text("{}\n") + (self.repo / "policy.json").write_text("{}\n") + subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + self.task["project"] = { + "repo_path": str(self.repo), "base_ref": "HEAD", "target_ref": "HEAD", + } + self.task["routing"]["profiles_file"] = "profiles.json" + self.task["routing"]["policy_file"] = "policy.json" + self.context = multiprocessing.get_context("spawn") + + def run_processes(self, targets): + results = self.context.Queue() + barrier = self.context.Barrier(len(targets)) + processes = [self.context.Process(target=target, args=(*args, barrier, results)) for target,args in targets] + try: + for process in processes: + process.start() + outcomes = [results.get(timeout=20) for _ in processes] + for process in processes: + process.join(timeout=5) + self.assertTrue(all(process.exitcode == 0 for process in processes), processes) + return outcomes + finally: + for process in processes: + if process.is_alive(): + process.terminate() + process.join(timeout=2) + results.close() + results.join_thread() + + def test_identical_public_starts_share_one_run_across_processes(self): + args = (str(self.runtime), self.task, "same-process-key") + outcomes = self.run_processes([(service_start,args),(service_start,args)]) + self.assertTrue(all(outcome[0] == "ok" for outcome in outcomes), outcomes) + self.assertEqual(len({outcome[1] for outcome in outcomes}), 1) + self.assertEqual(sorted(outcome[2] for outcome in outcomes), [False, True]) + run_id = outcomes[0][1] + result = Service(self.runtime).result(run_id) + self.assertTrue(result["ready"]) + self.assertEqual([item["name"] for item in result["artifacts"]], ["result-receipt.json"]) + + def test_changed_body_conflicts_with_same_key_across_processes(self): + changed = json.loads(json.dumps(self.task)) + changed["goal"] += " changed" + outcomes = self.run_processes([ + (service_start,(str(self.runtime),self.task,"conflicting-process-key")), + (service_start,(str(self.runtime),changed,"conflicting-process-key")), + ]) + self.assertEqual(sorted(outcome[0] for outcome in outcomes), ["error", "ok"]) + error = next(outcome for outcome in outcomes if outcome[0] == "error") + self.assertEqual(error[1], "ConflictError") + + def test_two_processes_cannot_reserve_two_worktree_writers(self): + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + ready = [] + for key in ("writer-a", "writer-b"): + claim = store.claim_start(self.repo, key, {"task": key}, "preflight") + version = store.complete_preparation(claim.run_id, claim.fencing_token, {"head": key}) + ready.append((claim.run_id, version)) + finally: + store.close() + common = (str(self.runtime / "state.sqlite3"), str(self.runtime / "artifacts")) + outcomes = self.run_processes([ + (reserve_writer,(*common,*ready[0],"owner-a")), + (reserve_writer,(*common,*ready[1],"owner-b")), + ]) + self.assertEqual(sorted(outcome[0] for outcome in outcomes), ["error", "ok"]) + self.assertEqual(next(item for item in outcomes if item[0] == "error")[2], "ConflictError") + winner = next(item for item in outcomes if item[0] == "ok")[1] + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + store.cancel_launching(winner) + finally: + store.close() + + def test_repeated_cancel_is_idempotent_across_processes(self): + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + claim = store.claim_start(self.repo, "cancel-race", {"task": 1}, "preflight") + store.complete_preparation(claim.run_id, claim.fencing_token, {"head": "fixed"}) + finally: + store.close() + args = (str(self.runtime), claim.run_id) + outcomes = self.run_processes([(service_cancel,args),(service_cancel,args)]) + self.assertTrue(all(outcome[0:2] == ("ok", "cancelled") for outcome in outcomes), outcomes) + self.assertEqual(len({outcome[2] for outcome in outcomes}), 1) + result = Service(self.runtime).result(claim.run_id) + self.assertEqual([item["name"] for item in result["artifacts"]], ["result-receipt.json"]) + + +if __name__ == "__main__": + unittest.main() From 2c3e9d76c5972e973be93cb6a5dd4f430656a3b9 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 15 Sep 2026 11:19:42 +0530 Subject: [PATCH 041/197] feat: add fenced host handoff storage --- .../devsquad/migrations/005_host_handoffs.sql | 55 ++ plugin/core/src/devsquad/store.py | 574 +++++++++++++++++- test/core/test_cli.py | 7 +- test/core/test_handoff_store.py | 494 +++++++++++++++ test/core/test_store.py | 8 +- 5 files changed, 1128 insertions(+), 10 deletions(-) create mode 100644 plugin/core/src/devsquad/migrations/005_host_handoffs.sql create mode 100644 test/core/test_handoff_store.py diff --git a/plugin/core/src/devsquad/migrations/005_host_handoffs.sql b/plugin/core/src/devsquad/migrations/005_host_handoffs.sql new file mode 100644 index 0000000..9396afb --- /dev/null +++ b/plugin/core/src/devsquad/migrations/005_host_handoffs.sql @@ -0,0 +1,55 @@ +CREATE TABLE handoffs ( + id TEXT PRIMARY KEY, + run_id TEXT NOT NULL REFERENCES runs(id), + sequence INTEGER NOT NULL CHECK(sequence > 0), + packet_json TEXT NOT NULL, + packet_sha256 TEXT NOT NULL + CHECK(length(packet_sha256) = 64 AND packet_sha256 NOT GLOB '*[^0-9a-f]*'), + status TEXT NOT NULL CHECK(status IN ('open', 'submitted', 'consumed', 'cancelled')), + created_run_version INTEGER NOT NULL CHECK(created_run_version > 0), + submitted_run_version INTEGER, + created_at TEXT NOT NULL, + closed_at TEXT, + UNIQUE(run_id, sequence) +); + +CREATE UNIQUE INDEX one_pending_handoff_per_run +ON handoffs(run_id) +WHERE status IN ('open', 'submitted'); + +ALTER TABLE claims ADD COLUMN handoff_id TEXT REFERENCES handoffs(id); +ALTER TABLE claims ADD COLUMN lease_expires_at TEXT; +ALTER TABLE claims ADD COLUMN renewed_at TEXT; + +CREATE UNIQUE INDEX one_current_claim_per_handoff +ON claims(handoff_id) +WHERE handoff_id IS NOT NULL; + +CREATE TABLE handoff_submissions ( + id TEXT PRIMARY KEY, + handoff_id TEXT NOT NULL REFERENCES handoffs(id), + submission_id TEXT NOT NULL, + submission_hash TEXT NOT NULL + CHECK(length(submission_hash) = 64 AND submission_hash NOT GLOB '*[^0-9a-f]*'), + owner_id TEXT NOT NULL, + fencing_token INTEGER NOT NULL CHECK(fencing_token > 0), + disposition TEXT NOT NULL CHECK(disposition IN ('accept', 'revise', 'reject')), + decision_json TEXT NOT NULL, + evidence_refs_json TEXT NOT NULL, + outcome TEXT NOT NULL CHECK(outcome IN ('recorded', 'rejected')), + rejection_code TEXT, + recorded_run_version INTEGER NOT NULL CHECK(recorded_run_version > 0), + created_at TEXT NOT NULL, + CHECK( + (outcome = 'recorded' AND rejection_code IS NULL) + OR (outcome = 'rejected' AND rejection_code IS NOT NULL) + ), + UNIQUE(handoff_id, submission_id, submission_hash) +); + +CREATE UNIQUE INDEX one_recorded_submission_per_handoff +ON handoff_submissions(handoff_id) +WHERE outcome = 'recorded'; + +CREATE INDEX handoff_submission_id_lookup +ON handoff_submissions(handoff_id, submission_id); diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index d449416..c8e1e4a 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -3,7 +3,7 @@ from __future__ import annotations from dataclasses import dataclass -from datetime import datetime, timezone +from datetime import datetime, timedelta, timezone import hashlib from importlib.resources import files import json @@ -17,8 +17,9 @@ from .contracts import ContractError -SUPPORTED_SCHEMA_VERSION = 4 +SUPPORTED_SCHEMA_VERSION = 5 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} +HOST_LEASE_SECONDS = 10 * 60 class ConflictError(ContractError): @@ -55,10 +56,67 @@ class PreparationClaim: version: int +@dataclass(frozen=True) +class HandoffClaim: + run_id: str + handoff_id: str + owner_id: str + fencing_token: int + expires_at: str + run_version: int + action: str + + +@dataclass(frozen=True) +class HandoffSnapshot: + run_id: str + run_state: str + run_phase: str | None + run_version: int + handoff_id: str + sequence: int + status: str + packet: dict[str, Any] + packet_sha256: str + created_run_version: int + submitted_run_version: int | None + created_at: str + closed_at: str | None + claim: HandoffClaim | None + + +@dataclass(frozen=True) +class HandoffSubmission: + run_id: str + handoff_id: str + submission_id: str + submission_hash: str + disposition: str + recorded_run_version: int + replayed: bool + + def _utc_now() -> str: return datetime.now(timezone.utc).isoformat() +def _authoritative_now(value: datetime | None = None) -> datetime: + current = datetime.now(timezone.utc) if value is None else value + if current.tzinfo is None or current.utcoffset() is None: + raise ContractError("authoritative time must include a timezone") + return current.astimezone(timezone.utc) + + +def _parse_utc(value: str) -> datetime: + try: + parsed = datetime.fromisoformat(value) + except (TypeError, ValueError) as exc: + raise ConflictError("persisted handoff lease timestamp is invalid") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ConflictError("persisted handoff lease timestamp is invalid") + return parsed.astimezone(timezone.utc) + + def canonical_json(value: Any) -> str: try: return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False) @@ -243,8 +301,16 @@ def validate_predecessor(self, run_id: str, fencing_token: int, supersedes_run_i or row["predecessor_state"] not in TERMINAL_STATES): raise ConflictError("superseded run must be terminal and belong to the same project") - def _terminal_receipt(self, run_id: str, terminal_state: str, phase: str, payload: Any) -> tuple[Path, str, int, str]: - finished_at = _utc_now() + def _terminal_receipt( + self, + run_id: str, + terminal_state: str, + phase: str, + payload: Any, + *, + now: str | None = None, + ) -> tuple[Path, str, int, str]: + finished_at = now or _utc_now() receipt = { "schema_version": 1, "run_id": run_id, @@ -813,6 +879,506 @@ def active_attempt(self, run_id: str) -> dict[str, Any] | None: row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND status IN ('reserved','running','cancelling','ownership_ambiguous') ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() return dict(row) if row else None + @staticmethod + def _handoff_claim_from_row(row: sqlite3.Row, action: str) -> HandoffClaim: + return HandoffClaim( + run_id=row["run_id"], + handoff_id=row["handoff_id"], + owner_id=row["owner_id"], + fencing_token=row["fencing_token"], + expires_at=row["lease_expires_at"], + run_version=row["run_version"], + action=action, + ) + + def publish_handoff( + self, + run_id: str, + expected_version: int, + attempt_token: str, + supervisor_token: int, + packet: dict[str, Any], + *, + now: datetime | None = None, + ) -> HandoffSnapshot: + """Release a reconciled writer and publish one immutable host packet. + + Process absence is established by the owning supervisor before it calls + this storage transition, just as it is before an ordinary attempt + completion. The persisted attempt and supervisor fences ensure a stale + coordinator cannot publish after ownership has moved. + """ + if (type(expected_version) is not int or expected_version < 1 + or not isinstance(attempt_token, str) or not attempt_token + or type(supervisor_token) is not int + or supervisor_token < 1 or not isinstance(packet, dict)): + raise ContractError("handoff publication requires valid ownership and packet fields") + packet_json = canonical_json(packet) + packet_sha256 = hashlib.sha256(packet_json.encode()).hexdigest() + timestamp = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,a.id AS attempt_id,a.status AS attempt_status," + "s.fencing_token AS supervisor_token,s.active AS supervisor_active " + "FROM runs r JOIN attempts a ON a.run_id=r.id " + "JOIN supervisor_claims s ON s.run_id=r.id " + "WHERE r.id=? AND a.attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if (not row or row["state"] != "running" or row["phase"] is not None + or row["version"] != expected_version + or row["attempt_status"] != "running" + or not row["supervisor_active"] + or row["supervisor_token"] != supervisor_token): + raise ConflictError("handoff publication is fenced") + if self.connection.execute( + "SELECT 1 FROM handoffs WHERE run_id=? AND status IN ('open','submitted')", + (run_id,), + ).fetchone(): + raise ConflictError("run already has a pending handoff") + sequence = self.connection.execute( + "SELECT COALESCE(MAX(sequence),0)+1 FROM handoffs WHERE run_id=?", (run_id,), + ).fetchone()[0] + handoff_id, version = str(uuid.uuid4()), expected_version + 1 + self.connection.execute( + "INSERT INTO handoffs(id,run_id,sequence,packet_json,packet_sha256,status," + "created_run_version,created_at) VALUES(?,?,?,?,?,'open',?,?)", + (handoff_id, run_id, sequence, packet_json, packet_sha256, version, timestamp), + ) + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (timestamp, row["attempt_id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=? AND fencing_token=?", + (run_id, supervisor_token), + ) + self.connection.execute( + "UPDATE runs SET state='awaiting_host',phase=NULL,version=?,updated_at=? WHERE id=?", + (version, timestamp, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff_id, + "packet_sha256": packet_sha256, + "sequence": sequence, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.awaiting_host',?,?)", + (run_id, version, payload, timestamp), + ) + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + snapshot = self.handoff_snapshot(run_id) + if snapshot is None: # Defensive: the just-committed projection must exist. + raise ConflictError("published handoff is missing") + return snapshot + + def handoff_snapshot(self, run_id: str) -> HandoffSnapshot | None: + self.connection.execute("BEGIN") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if not run: + raise ContractError("run does not exist") + handoff = self.connection.execute( + "SELECT * FROM handoffs WHERE run_id=? ORDER BY sequence DESC LIMIT 1", (run_id,), + ).fetchone() + claim = None + if handoff: + claim_row = self.connection.execute( + "SELECT run_id,handoff_id,owner_id,fencing_token,lease_expires_at," + "? AS run_version FROM claims WHERE run_id=? AND kind='host' " + "AND handoff_id=? AND active=1", + (run["version"], run_id, handoff["id"]), + ).fetchone() + if claim_row: + claim = self._handoff_claim_from_row(claim_row, "current") + if not handoff: + self.connection.execute("COMMIT") + return None + packet_json = handoff["packet_json"] + if hashlib.sha256(packet_json.encode()).hexdigest() != handoff["packet_sha256"]: + raise ConflictError("persisted handoff packet hash does not match") + snapshot = HandoffSnapshot( + run_id=run_id, + run_state=run["state"], + run_phase=run["phase"], + run_version=run["version"], + handoff_id=handoff["id"], + sequence=handoff["sequence"], + status=handoff["status"], + packet=json.loads(packet_json), + packet_sha256=handoff["packet_sha256"], + created_run_version=handoff["created_run_version"], + submitted_run_version=handoff["submitted_run_version"], + created_at=handoff["created_at"], + closed_at=handoff["closed_at"], + claim=claim, + ) + self.connection.execute("COMMIT") + return snapshot + except Exception: + self.connection.execute("ROLLBACK") + raise + + def claim_handoff( + self, + run_id: str, + expected_version: int, + owner_id: str, + prior_claim: HandoffClaim | None = None, + *, + now: datetime | None = None, + ) -> HandoffClaim: + if (type(expected_version) is not int or expected_version < 1 + or not isinstance(owner_id, str) or not owner_id): + raise ContractError("handoff claim requires a run version and owner") + if prior_claim is not None and not isinstance(prior_claim, HandoffClaim): + raise ContractError("prior handoff claim is invalid") + current = _authoritative_now(now) + timestamp = current.isoformat() + expires_at = (current + timedelta(seconds=HOST_LEASE_SECONDS)).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,h.id AS handoff_id,h.status," + "c.kind,c.owner_id,c.fencing_token,c.active,c.handoff_id AS claim_handoff_id," + "c.lease_expires_at " + "FROM runs r JOIN handoffs h ON h.run_id=r.id " + "JOIN claims c ON c.run_id=r.id " + "WHERE r.id=? ORDER BY h.sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + if not row: + raise ContractError("run has no handoff") + if (row["state"] != "awaiting_host" or row["phase"] is not None + or row["status"] != "open" or row["version"] != expected_version): + raise ConflictError("handoff is not claimable at that run version") + live = bool( + row["kind"] == "host" and row["active"] + and row["claim_handoff_id"] == row["handoff_id"] + and row["lease_expires_at"] + and current < _parse_utc(row["lease_expires_at"]) + ) + if prior_claim is not None: + if (not live or prior_claim.run_id != run_id + or prior_claim.handoff_id != row["handoff_id"] + or prior_claim.owner_id != owner_id + or row["owner_id"] != owner_id + or prior_claim.fencing_token != row["fencing_token"]): + raise ConflictError("handoff renewal claim is stale or expired") + token, action = row["fencing_token"], "renewed" + self.connection.execute( + "UPDATE claims SET lease_expires_at=?,renewed_at=? WHERE run_id=?", + (expires_at, timestamp, run_id), + ) + else: + if live: + raise ConflictError("handoff already has a live claim") + prior_host_claim = row["kind"] == "host" and row["claim_handoff_id"] == row["handoff_id"] + token = row["fencing_token"] + 1 + action = "taken_over" if prior_host_claim else "acquired" + self.connection.execute( + "UPDATE claims SET kind='host',fencing_token=?,owner_id=?,active=1," + "claimed_at=?,handoff_id=?,lease_expires_at=?,renewed_at=NULL WHERE run_id=?", + (token, owner_id, timestamp, row["handoff_id"], expires_at, run_id), + ) + version = expected_version + 1 + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, timestamp, run_id), + ) + payload = canonical_json({ + "action": action, + "expires_at": expires_at, + "fencing_token": token, + "handoff_id": row["handoff_id"], + "owner_id": owner_id, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", + (run_id, version, f"handoff.{action}", payload, timestamp), + ) + self.connection.execute("COMMIT") + return HandoffClaim( + run_id=run_id, + handoff_id=row["handoff_id"], + owner_id=owner_id, + fencing_token=token, + expires_at=expires_at, + run_version=version, + action=action, + ) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def _validated_handoff_decision( + self, run_id: str, decision: dict[str, Any], + ) -> tuple[str, str, str, str, list[dict[str, str]]]: + expected_fields = { + "schema_version", "submission_id", "submission_hash", + "disposition", "reason", "evidence_refs", + } + if not isinstance(decision, dict) or set(decision) != expected_fields: + raise ContractError("handoff decision fields are invalid") + if decision["schema_version"] != 1 or type(decision["schema_version"]) is not int: + raise ContractError("handoff decision schema_version is invalid") + submission_id = decision["submission_id"] + disposition = decision["disposition"] + reason = decision["reason"] + evidence_refs = decision["evidence_refs"] + if not isinstance(submission_id, str) or not submission_id: + raise ContractError("handoff submission_id is required") + if disposition not in {"accept", "revise", "reject"}: + raise ContractError("handoff disposition is invalid") + if not isinstance(reason, str) or (disposition == "revise" and not reason): + raise ContractError("handoff decision reason is invalid") + if not isinstance(evidence_refs, list): + raise ContractError("handoff evidence_refs must be an array") + normalized_refs = [] + for evidence in evidence_refs: + if (not isinstance(evidence, dict) or set(evidence) != {"artifact_id", "sha256"} + or not isinstance(evidence["artifact_id"], str) + or not evidence["artifact_id"] + or not isinstance(evidence["sha256"], str)): + raise ContractError("handoff evidence reference is invalid") + artifact = self.connection.execute( + "SELECT sha256 FROM artifacts WHERE id=? AND run_id=?", + (evidence["artifact_id"], run_id), + ).fetchone() + if not artifact or artifact["sha256"] != evidence["sha256"]: + raise ContractError("handoff evidence does not match a run artifact") + normalized_refs.append({ + "artifact_id": evidence["artifact_id"], "sha256": evidence["sha256"], + }) + body = {key: decision[key] for key in expected_fields if key != "submission_hash"} + expected_hash = request_hash(body) + if decision["submission_hash"] != expected_hash: + raise ContractError("handoff submission hash does not match the decision") + return submission_id, expected_hash, disposition, reason, normalized_refs + + def record_handoff_submission( + self, + run_id: str, + claim: HandoffClaim, + decision: dict[str, Any], + *, + now: datetime | None = None, + ) -> HandoffSubmission: + if not isinstance(claim, HandoffClaim) or claim.run_id != run_id: + raise ContractError("handoff completion claim is invalid") + submission_id, submission_hash, disposition, _, evidence_refs = ( + self._validated_handoff_decision(run_id, decision) + ) + decision_json = canonical_json(decision) + evidence_json = canonical_json(evidence_refs) + current = _authoritative_now(now) + timestamp = current.isoformat() + rejection = None + result = None + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if not run: + raise ContractError("run does not exist") + handoff = self.connection.execute( + "SELECT * FROM handoffs WHERE id=? AND run_id=?", (claim.handoff_id, run_id), + ).fetchone() + if not handoff: + raise ConflictError("handoff completion targets a different run") + prior = self.connection.execute( + "SELECT * FROM handoff_submissions WHERE handoff_id=? AND submission_id=? " + "AND submission_hash=? ORDER BY created_at LIMIT 1", + (claim.handoff_id, submission_id, submission_hash), + ).fetchone() + if prior: + if prior["outcome"] == "recorded": + self.connection.execute("COMMIT") + return HandoffSubmission( + run_id=run_id, + handoff_id=claim.handoff_id, + submission_id=submission_id, + submission_hash=submission_hash, + disposition=prior["disposition"], + recorded_run_version=prior["recorded_run_version"], + replayed=True, + ) + rejection = prior["rejection_code"] or "handoff_submission_rejected" + reused_id = self.connection.execute( + "SELECT 1 FROM handoff_submissions WHERE handoff_id=? AND submission_id=? " + "AND submission_hash<>?", + (claim.handoff_id, submission_id, submission_hash), + ).fetchone() + if reused_id and rejection is None: + rejection = "submission_id_reused" + persisted_claim = self.connection.execute( + "SELECT kind,owner_id,fencing_token,active,handoff_id,lease_expires_at " + "FROM claims WHERE run_id=?", (run_id,), + ).fetchone() + if rejection is None: + if run["state"] in TERMINAL_STATES: + rejection = "terminal_run" + elif (run["state"] != "awaiting_host" or run["phase"] is not None + or handoff["status"] != "open"): + rejection = "handoff_not_open" + elif (not persisted_claim or persisted_claim["kind"] != "host" + or not persisted_claim["active"] + or persisted_claim["handoff_id"] != claim.handoff_id + or persisted_claim["owner_id"] != claim.owner_id + or persisted_claim["fencing_token"] != claim.fencing_token): + rejection = "stale_claim" + elif (not persisted_claim["lease_expires_at"] + or current >= _parse_utc(persisted_claim["lease_expires_at"])): + rejection = "expired_claim" + if rejection is None: + version = run["version"] + 1 + submission_row_id = str(uuid.uuid4()) + self.connection.execute( + "INSERT INTO handoff_submissions(id,handoff_id,submission_id,submission_hash," + "owner_id,fencing_token,disposition,decision_json,evidence_refs_json,outcome," + "rejection_code,recorded_run_version,created_at) " + "VALUES(?,?,?,?,?,?,?,?,?,'recorded',NULL,?,?)", + (submission_row_id, claim.handoff_id, submission_id, submission_hash, + claim.owner_id, claim.fencing_token, disposition, decision_json, + evidence_json, version, timestamp), + ) + self.connection.execute( + "UPDATE handoffs SET status='submitted',submitted_run_version=?,closed_at=? " + "WHERE id=?", (version, timestamp, claim.handoff_id), + ) + self.connection.execute( + "UPDATE claims SET active=0 WHERE run_id=? AND handoff_id=?", + (run_id, claim.handoff_id), + ) + self.connection.execute( + "UPDATE runs SET phase='handoff_submitted',version=?,updated_at=? WHERE id=?", + (version, timestamp, run_id), + ) + payload = canonical_json({ + "disposition": disposition, + "handoff_id": claim.handoff_id, + "submission_hash": submission_hash, + "submission_id": submission_id, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'handoff.submitted',?,?)", + (run_id, version, payload, timestamp), + ) + result = HandoffSubmission( + run_id=run_id, + handoff_id=claim.handoff_id, + submission_id=submission_id, + submission_hash=submission_hash, + disposition=disposition, + recorded_run_version=version, + replayed=False, + ) + else: + existing_rejection = self.connection.execute( + "SELECT recorded_run_version FROM handoff_submissions WHERE handoff_id=? " + "AND submission_id=? AND submission_hash=? AND outcome='rejected'", + (claim.handoff_id, submission_id, submission_hash), + ).fetchone() + if not existing_rejection: + terminal_audit = run["state"] in TERMINAL_STATES + version = run["version"] if terminal_audit else run["version"] + 1 + self.connection.execute( + "INSERT INTO handoff_submissions(id,handoff_id,submission_id,submission_hash," + "owner_id,fencing_token,disposition,decision_json,evidence_refs_json,outcome," + "rejection_code,recorded_run_version,created_at) " + "VALUES(?,?,?,?,?,?,?,?,?,'rejected',?,?,?)", + (str(uuid.uuid4()), claim.handoff_id, submission_id, submission_hash, + claim.owner_id, claim.fencing_token, disposition, decision_json, + evidence_json, rejection, version, timestamp), + ) + if not terminal_audit: + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", + (version, timestamp, run_id), + ) + payload = canonical_json({ + "handoff_id": claim.handoff_id, + "reason": rejection, + "submission_hash": submission_hash, + "submission_id": submission_id, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'handoff.completion_rejected',?,?)", + (run_id, version, payload, timestamp), + ) + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + if rejection is not None: + raise ConflictError(f"handoff completion rejected: {rejection}") + if result is None: + raise ConflictError("handoff completion was not recorded") + return result + + def cancel_host_wait(self, run_id: str, *, now: datetime | None = None) -> int: + current = _authoritative_now(now) + timestamp = current.isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if not run: + raise ContractError("run does not exist") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"] + handoff = self.connection.execute( + "SELECT id,status FROM handoffs WHERE run_id=? ORDER BY sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + if (run["state"] != "awaiting_host" or not handoff + or handoff["status"] not in {"open", "submitted"} + or run["phase"] not in {None, "handoff_submitted"}): + raise ConflictError("run is not awaiting a cancellable host handoff") + path, digest, size, receipt_time = self._terminal_receipt( + run_id, "cancelled", "awaiting_host", None, now=timestamp, + ) + version = self._reference_terminal_receipt( + run_id, run["version"], path, digest, size, receipt_time, + ) + 1 + self.connection.execute( + "UPDATE handoffs SET status='cancelled',closed_at=? WHERE id=?", + (timestamp, handoff["id"]), + ) + self.connection.execute( + "UPDATE claims SET active=0,fencing_token=fencing_token+1 " + "WHERE run_id=? AND kind='host' AND handoff_id=?", + (run_id, handoff["id"]), + ) + self.connection.execute( + "UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?", + (version, timestamp, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff["id"], "receipt": "result-receipt.json", + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, payload, timestamp), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + def cancel_queued(self, run_id: str) -> int: self.connection.execute("BEGIN IMMEDIATE") try: diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 08d8c7e..5e480bd 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -230,7 +230,7 @@ def build_python(): return candidate return None - def test_installed_wheel_contains_and_applies_migration_four(self): + def test_installed_wheel_contains_and_applies_migrations_through_five(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") @@ -272,9 +272,12 @@ def test_installed_wheel_contains_and_applies_migration_four(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 4 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 5 columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} assert {"package_path", "package_digest", "supersedes_run_id"} <= columns + assert store.connection.execute( + "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoffs'" + ).fetchone() finally: store.close() ''' diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py new file mode 100644 index 0000000..9e9ddfe --- /dev/null +++ b/test/core/test_handoff_store.py @@ -0,0 +1,494 @@ +from datetime import datetime, timedelta, timezone +import json +import os +from pathlib import Path +import shutil +import sqlite3 +import subprocess +import sys +import tempfile +import threading +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.store import ( + ConflictError, + HandoffClaim, + Store, + request_hash, +) + + +class HandoffStoreTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-handoff-") + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True, + ) + (self.repo / "README").write_text("base\n") + subprocess.run(["git", "-C", str(self.repo), "add", "README"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.database = self.root / "runtime/state.sqlite3" + self.artifacts = self.root / "runtime/artifacts" + self.store = Store(self.database, self.artifacts) + + def tearDown(self): + self.store.close() + self.temp.cleanup() + + def running_run(self, key="handoff"): + claim = self.store.claim_start(self.repo, key, {"task": key}, "preflight") + version = self.store.complete_preparation( + claim.run_id, + claim.fencing_token, + {"base_oid": "a" * 40, "target_oid": "b" * 40}, + package_path="/frozen/package", + package_digest="package-digest", + ) + reservation = self.store.reserve_attempt( + claim.run_id, version, "supervisor", "package-digest", + ) + version = self.store.mark_attempt_running( + reservation, 101, 101, "process-start-id", + ) + return claim.run_id, reservation, version + + def waiting_run(self, key="handoff"): + run_id, reservation, version = self.running_run(key) + packet = { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "evidence_refs": [], + } + snapshot = self.store.publish_handoff( + run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + packet, + now=datetime(2026, 9, 15, 5, 0, tzinfo=timezone.utc), + ) + return run_id, reservation, snapshot + + @staticmethod + def decision(submission_id="submission-1", disposition="accept", reason="accepted"): + body = { + "schema_version": 1, + "submission_id": submission_id, + "disposition": disposition, + "reason": reason, + "evidence_refs": [], + } + return {**body, "submission_hash": request_hash(body)} + + def test_schema_four_fixture_migrates_to_host_handoffs(self): + old_database = self.root / "schema-four.sqlite3" + connection = sqlite3.connect(old_database) + migration_dir = ROOT / "plugin/core/src/devsquad/migrations" + for version, name in ( + (1, "001_initial.sql"), + (2, "002_supervisor.sql"), + (3, "003_durable_io.sql"), + (4, "004_run_snapshot.sql"), + ): + connection.executescript((migration_dir / name).read_text()) + connection.execute( + "INSERT INTO schema_migrations(version,applied_at) VALUES(?, 'fixture')", + (version,), + ) + connection.commit() + connection.close() + + upgraded = Store(old_database, self.root / "schema-four-artifacts") + self.addCleanup(upgraded.close) + self.assertEqual( + upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], + 5, + ) + tables = { + row[0] + for row in upgraded.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table'" + ) + } + self.assertTrue({"handoffs", "handoff_submissions"} <= tables) + claim_columns = { + row[1] for row in upgraded.connection.execute("PRAGMA table_info(claims)") + } + self.assertTrue({"handoff_id", "lease_expires_at", "renewed_at"} <= claim_columns) + with self.assertRaises(sqlite3.IntegrityError): + upgraded.connection.execute( + "INSERT INTO handoffs(id,run_id,sequence,packet_json,packet_sha256,status," + "created_run_version,created_at) VALUES('bad','missing',1,'{}',?,'invalid',1,'now')", + ("0" * 64,), + ) + + def test_publish_is_fenced_and_atomically_releases_writer(self): + run_id, reservation, version = self.running_run() + packet = {"schema_version": 1, "candidate_sha256": "c" * 64} + snapshot = self.store.publish_handoff( + run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + packet, + ) + self.assertEqual(snapshot.packet, packet) + self.assertEqual(snapshot.run_state, "awaiting_host") + self.assertEqual(snapshot.run_version, version + 1) + self.assertEqual(snapshot.created_run_version, version + 1) + self.assertEqual( + self.store.connection.execute( + "SELECT status FROM attempts WHERE attempt_token=?", + (reservation.attempt_token,), + ).fetchone()[0], + "finished", + ) + self.assertEqual( + self.store.connection.execute( + "SELECT active FROM supervisor_claims WHERE run_id=?", (run_id,), + ).fetchone()[0], + 0, + ) + event = self.store.connection.execute( + "SELECT run_version,type,payload FROM events WHERE run_id=? ORDER BY id DESC LIMIT 1", + (run_id,), + ).fetchone() + self.assertEqual((event[0], event[1]), (version + 1, "run.awaiting_host")) + self.assertEqual(json.loads(event[2])["packet_sha256"], snapshot.packet_sha256) + with self.assertRaises(ConflictError): + self.store.publish_handoff( + run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + packet, + ) + self.store.connection.execute( + "UPDATE handoffs SET packet_json='{}' WHERE id=?", (snapshot.handoff_id,), + ) + with self.assertRaises(ConflictError): + self.store.handoff_snapshot(run_id) + + def test_claim_cas_renewal_and_expired_takeover(self): + run_id, _, snapshot = self.waiting_run() + first_time = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) + first = self.store.claim_handoff( + run_id, snapshot.run_version, "host-a", now=first_time, + ) + self.assertEqual(first.action, "acquired") + renewed = self.store.claim_handoff( + run_id, + first.run_version, + "host-a", + first, + now=first_time + timedelta(minutes=5), + ) + self.assertEqual(renewed.action, "renewed") + self.assertEqual(renewed.fencing_token, first.fencing_token) + self.assertGreater(renewed.expires_at, first.expires_at) + with self.assertRaises(ConflictError): + self.store.claim_handoff( + run_id, + renewed.run_version, + "host-b", + now=first_time + timedelta(minutes=6), + ) + with self.assertRaises(ConflictError): + self.store.claim_handoff( + run_id, + renewed.run_version, + "host-a", + renewed, + now=first_time + timedelta(minutes=15), + ) + takeover = self.store.claim_handoff( + run_id, + renewed.run_version, + "host-b", + now=first_time + timedelta(minutes=15), + ) + self.assertEqual(takeover.action, "taken_over") + self.assertEqual(takeover.fencing_token, first.fencing_token + 1) + + def test_two_claimants_at_one_version_yield_one_owner(self): + run_id, _, snapshot = self.waiting_run() + barrier = threading.Barrier(2) + claims = [] + conflicts = [] + + def claim(owner): + store = Store(self.database, self.artifacts) + try: + barrier.wait() + claims.append(store.claim_handoff(run_id, snapshot.run_version, owner)) + except ConflictError as exc: + conflicts.append(exc) + finally: + store.close() + + threads = [threading.Thread(target=claim, args=(owner,)) for owner in ("a", "b")] + for thread in threads: + thread.start() + for thread in threads: + thread.join() + self.assertEqual(len(claims), 1) + self.assertEqual(len(conflicts), 1) + + def test_expired_submission_is_audited_and_cannot_displace_takeover(self): + run_id, _, snapshot = self.waiting_run() + started = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) + old_claim = self.store.claim_handoff( + run_id, snapshot.run_version, "old-host", now=started, + ) + takeover = self.store.claim_handoff( + run_id, + old_claim.run_version, + "new-host", + now=started + timedelta(minutes=11), + ) + before = self.store.run(run_id)["version"] + with self.assertRaises(ConflictError): + self.store.record_handoff_submission( + run_id, + old_claim, + self.decision(), + now=started + timedelta(minutes=11), + ) + self.assertEqual(self.store.run(run_id)["version"], before + 1) + rejection = self.store.connection.execute( + "SELECT outcome,rejection_code FROM handoff_submissions WHERE handoff_id=?", + (snapshot.handoff_id,), + ).fetchone() + self.assertEqual(tuple(rejection), ("rejected", "stale_claim")) + current_claim = self.store.handoff_snapshot(run_id).claim + self.assertEqual(current_claim.owner_id, takeover.owner_id) + self.assertEqual(current_claim.fencing_token, takeover.fencing_token) + + def test_submission_replays_and_terminal_late_rejection_preserves_run_and_events(self): + run_id, _, snapshot = self.waiting_run() + claimed_at = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) + claim = self.store.claim_handoff( + run_id, snapshot.run_version, "host", now=claimed_at, + ) + decision = self.decision() + recorded = self.store.record_handoff_submission( + run_id, claim, decision, now=claimed_at + timedelta(minutes=1), + ) + self.assertFalse(recorded.replayed) + waiting = self.store.run(run_id) + self.assertEqual((waiting["state"], waiting["phase"]), ("awaiting_host", "handoff_submitted")) + replay = self.store.record_handoff_submission( + run_id, claim, decision, now=claimed_at + timedelta(minutes=20), + ) + self.assertTrue(replay.replayed) + self.assertEqual(replay.recorded_run_version, recorded.recorded_run_version) + terminal_version = self.store.cancel_host_wait( + run_id, now=claimed_at + timedelta(minutes=21), + ) + replay_after_cancel = self.store.record_handoff_submission( + run_id, claim, decision, now=claimed_at + timedelta(minutes=22), + ) + self.assertTrue(replay_after_cancel.replayed) + self.assertEqual(self.store.run(run_id)["version"], terminal_version) + + terminal_before = self.store.connection.execute( + "SELECT * FROM runs WHERE id=?", (run_id,), + ).fetchone() + terminal_before = tuple(terminal_before) + events_before = [ + tuple(row) + for row in self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? ORDER BY id", (run_id,), + ) + ] + + conflicting = self.decision(reason="different body") + with self.assertRaises(ConflictError): + self.store.record_handoff_submission( + run_id, claim, conflicting, now=claimed_at + timedelta(minutes=23), + ) + terminal = self.store.run(run_id) + self.assertEqual(terminal["state"], "cancelled") + self.assertEqual(terminal["phase"], None) + self.assertEqual(terminal["version"], terminal_version) + terminal_after = tuple(self.store.connection.execute( + "SELECT * FROM runs WHERE id=?", (run_id,), + ).fetchone()) + events_after = [ + tuple(row) + for row in self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? ORDER BY id", (run_id,), + ) + ] + self.assertEqual(terminal_after, terminal_before) + self.assertEqual(events_after, events_before) + rejection = self.store.connection.execute( + "SELECT outcome,rejection_code,recorded_run_version FROM handoff_submissions " + "WHERE handoff_id=? AND submission_hash=?", + (snapshot.handoff_id, conflicting["submission_hash"]), + ).fetchone() + self.assertEqual( + tuple(rejection), ("rejected", "submission_id_reused", terminal_version), + ) + + with self.assertRaises(ConflictError): + self.store.record_handoff_submission( + run_id, claim, conflicting, now=claimed_at + timedelta(minutes=24), + ) + self.assertEqual(tuple(self.store.connection.execute( + "SELECT * FROM runs WHERE id=?", (run_id,), + ).fetchone()), terminal_before) + self.assertEqual([ + tuple(row) + for row in self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? ORDER BY id", (run_id,), + ) + ], events_before) + + def test_cancel_wait_invalidates_claim_and_publishes_terminal_receipt(self): + run_id, _, snapshot = self.waiting_run() + claim = self.store.claim_handoff( + run_id, snapshot.run_version, "host", + now=datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc), + ) + version = self.store.cancel_host_wait( + run_id, now=datetime(2026, 9, 15, 5, 2, tzinfo=timezone.utc), + ) + run = self.store.run(run_id) + self.assertEqual((run["state"], run["phase"], run["version"]), ("cancelled", None, version)) + persisted_claim = self.store.connection.execute( + "SELECT active,fencing_token FROM claims WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(tuple(persisted_claim), (0, claim.fencing_token + 1)) + receipt = self.store.artifact_named(run_id, "result-receipt.json") + self.assertIsNotNone(receipt) + receipt_payload = json.loads(Path(receipt["path"]).read_text()) + self.assertEqual( + (receipt_payload["state"], receipt_payload["phase"]), + ("cancelled", "awaiting_host"), + ) + self.assertEqual(self.store.cancel_host_wait(run_id), version) + + +class InstalledWheelHandoffMigrationTest(unittest.TestCase): + @staticmethod + def build_python(): + candidates = [ + os.environ.get("DEVSQUAD_BUILD_PYTHON"), + sys.executable, + str(Path.home() / ".cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3"), + shutil.which("python3.13"), + shutil.which("python3.12"), + shutil.which("python3.11"), + ] + for candidate in dict.fromkeys(value for value in candidates if value): + result = subprocess.run( + [ + candidate, + "-c", + "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68", + ], + text=True, + capture_output=True, + ) + if result.returncode == 0: + return candidate + return None + + def test_installed_wheel_applies_schema_four_to_five(self): + build_python = self.build_python() + if build_python is None: + self.skipTest("offline wheel gate requires setuptools>=68 and wheel") + with tempfile.TemporaryDirectory(prefix="devsquad-handoff-wheel-") as directory: + root = Path(directory) + source = root / "core" + shutil.copytree(ROOT / "plugin/core", source) + wheels = root / "wheels" + wheels.mkdir() + subprocess.run( + [ + build_python, + "-m", + "pip", + "wheel", + str(source), + "--wheel-dir", + str(wheels), + "--no-index", + "--no-deps", + "--no-build-isolation", + ], + check=True, + text=True, + capture_output=True, + ) + wheel = next(wheels.glob("devsquad_core-*.whl")) + environment = os.environ.copy() + environment.pop("PYTHONPATH", None) + venv = root / "venv" + subprocess.run([build_python, "-m", "venv", str(venv)], check=True, env=environment) + python = venv / ("Scripts/python.exe" if os.name == "nt" else "bin/python") + subprocess.run( + [str(python), "-m", "pip", "install", "--no-index", "--no-deps", str(wheel)], + check=True, + text=True, + capture_output=True, + env=environment, + ) + probe = r''' +from importlib.resources import files +from pathlib import Path +import sqlite3 +import sys +from devsquad.store import Store + +root = Path(sys.argv[1]) +root.mkdir(parents=True) +database = root / "state.sqlite3" +connection = sqlite3.connect(database) +migrations = files("devsquad.migrations") +for version, name in ( + (1, "001_initial.sql"), + (2, "002_supervisor.sql"), + (3, "003_durable_io.sql"), + (4, "004_run_snapshot.sql"), +): + connection.executescript(migrations.joinpath(name).read_text()) + connection.execute( + "INSERT INTO schema_migrations(version, applied_at) VALUES(?, 'fixture')", (version,) + ) +connection.commit() +connection.close() +store = Store(database, root / "artifacts") +try: + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 5 + assert store.connection.execute( + "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" + ).fetchone() + claim_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(claims)")} + assert {"handoff_id", "lease_expires_at", "renewed_at"} <= claim_columns +finally: + store.close() +''' + subprocess.run( + [str(python), "-P", "-c", probe, str(root / "probe-runtime")], + check=True, + text=True, + capture_output=True, + cwd=root, + env=environment, + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_store.py b/test/core/test_store.py index bf45960..d16d1b8 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -174,8 +174,8 @@ def test_artifact_is_finalized_and_verified_before_reference(self): self.assertEqual(row[1], 13) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 4) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(5,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 5) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(6,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -190,7 +190,7 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 4) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 5) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) def test_version_three_fixture_adds_run_snapshot_columns(self): @@ -200,7 +200,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],4) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],5) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From a689b88139b9720fbc6c2a12cbf9d87dc32b034b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 15 Sep 2026 16:52:41 +0530 Subject: [PATCH 042/197] feat: expose fenced host handoffs --- plugin/core/src/devsquad/cli.py | 30 ++++ plugin/core/src/devsquad/service.py | 156 ++++++++++++++++++++- plugin/core/src/devsquad/store.py | 88 ++++++++---- test/core/test_cli.py | 61 ++++++++ test/core/test_handoff_service.py | 209 ++++++++++++++++++++++++++++ test/core/test_handoff_store.py | 8 ++ test/core/test_m2_cross_process.py | 84 ++++++++++- 7 files changed, 605 insertions(+), 31 deletions(-) create mode 100644 test/core/test_handoff_service.py diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 2f13328..018d7c5 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -125,6 +125,19 @@ def command_resume(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).resume(args.run, recovery)), 0 +def command_handoff_claim(args: argparse.Namespace) -> tuple[dict, int]: + prior_claim = _read_json(args.claim_file, "claim file") if args.claim_file else None + return envelope(data=_service(args).handoff_claim( + args.run, args.expected_version, args.owner, prior_claim, + )), 0 + + +def command_handoff_complete(args: argparse.Namespace) -> tuple[dict, int]: + claim = _read_json(args.claim_file, "claim file") + decision = _read_json(args.decision_file, "decision file") + return envelope(data=_service(args).handoff_complete(args.run, claim, decision)), 0 + + class ContractParser(argparse.ArgumentParser): def error(self, message: str) -> None: raise ContractError(message) @@ -159,6 +172,23 @@ def parser() -> argparse.ArgumentParser: if name == "resume": cmd.add_argument("--recovery-file") cmd.set_defaults(func=fn) events=sub.add_parser("events"); events.add_argument("run"); events.add_argument("--after",type=int,default=0); events.add_argument("--limit",type=int,default=100); events.add_argument("--json",action="store_true"); events.add_argument("--runtime-dir",default=runtime_default); events.set_defaults(func=command_events) + handoff = sub.add_parser("handoff") + handoff_sub = handoff.add_subparsers(dest="handoff_command", required=True) + claim = handoff_sub.add_parser("claim") + claim.add_argument("run") + claim.add_argument("--expected-version", type=int, required=True) + claim.add_argument("--owner", required=True) + claim.add_argument("--claim-file") + claim.add_argument("--json", action="store_true") + claim.add_argument("--runtime-dir", default=runtime_default) + claim.set_defaults(func=command_handoff_claim) + complete = handoff_sub.add_parser("complete") + complete.add_argument("run") + complete.add_argument("--claim-file", required=True) + complete.add_argument("--decision-file", required=True) + complete.add_argument("--json", action="store_true") + complete.add_argument("--runtime-dir", default=runtime_default) + complete.set_defaults(func=command_handoff_complete) return p diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 81dd21a..74b130e 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -4,6 +4,7 @@ import hashlib import json import os +from datetime import datetime from pathlib import Path import shutil import subprocess @@ -13,7 +14,14 @@ from typing import Any from .contracts import ContractError -from .store import ConflictError, Store, TERMINAL_STATES, canonical_json +from .store import ( + ConflictError, + HandoffClaim, + HandoffSnapshot, + Store, + TERMINAL_STATES, + canonical_json, +) from .validation import validate_task @@ -74,6 +82,70 @@ def _verified_package(self, run: dict[str, Any]) -> tuple[Path,str]: raise ConflictError("pinned package is missing or corrupt") return package,run["package_digest"] + @staticmethod + def _claim_payload(claim: HandoffClaim) -> dict[str, Any]: + return { + "schema_version": 1, + "run_id": claim.run_id, + "handoff_id": claim.handoff_id, + "owner": claim.owner_id, + "fencing_token": claim.fencing_token, + "expires_at": claim.expires_at, + "run_version": claim.run_version, + } + + @staticmethod + def _decode_claim(value: dict[str, Any]) -> HandoffClaim: + fields = { + "schema_version", "run_id", "handoff_id", "owner", + "fencing_token", "expires_at", "run_version", + } + if not isinstance(value, dict) or set(value) != fields: + raise ContractError("handoff claim fields are invalid") + if value["schema_version"] != 1 or type(value["schema_version"]) is not int: + raise ContractError("handoff claim schema_version is invalid") + for field in ("run_id", "handoff_id", "owner", "expires_at"): + if not isinstance(value[field], str) or not value[field]: + raise ContractError(f"handoff claim {field} is invalid") + for field in ("fencing_token", "run_version"): + if type(value[field]) is not int or value[field] < 1: + raise ContractError(f"handoff claim {field} is invalid") + try: + expires_at = datetime.fromisoformat(value["expires_at"]) + except ValueError as exc: + raise ContractError("handoff claim expires_at is invalid") from exc + if expires_at.tzinfo is None or expires_at.utcoffset() is None: + raise ContractError("handoff claim expires_at is invalid") + return HandoffClaim( + run_id=value["run_id"], + handoff_id=value["handoff_id"], + owner_id=value["owner"], + fencing_token=value["fencing_token"], + expires_at=value["expires_at"], + run_version=value["run_version"], + action="presented", + ) + + @staticmethod + def _handoff_payload(snapshot: HandoffSnapshot, *, include_packet: bool) -> dict[str, Any]: + payload = { + "handoff_id": snapshot.handoff_id, + "sequence": snapshot.sequence, + "status": snapshot.status, + "packet_sha256": snapshot.packet_sha256, + "created_run_version": snapshot.created_run_version, + "submitted_run_version": snapshot.submitted_run_version, + } + if include_packet: + payload["packet"] = snapshot.packet + if snapshot.claim is not None: + payload["claimed_by"] = snapshot.claim.owner_id + payload["claim_expires_at"] = snapshot.claim.expires_at + else: + payload["claimed_by"] = None + payload["claim_expires_at"] = None + return payload + @staticmethod def _resolve_snapshot( task: dict[str, Any], @@ -194,9 +266,28 @@ def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, diges def status(self, run_id: str) -> dict[str, Any]: store = self._store() try: - run, attempt = store.status_snapshot(run_id) + run, attempt, handoff = store.status_snapshot(run_id) active=attempt if attempt and attempt.get("status") in {"reserved","running","cancelling","ownership_ambiguous"} else None - return {"run_id": run_id, "state": run["state"], "phase": run["phase"], "version": run["version"], "active_attempt": {k: active.get(k) for k in ("id","status","pid","pgid","heartbeat_at")} if active else None, "next_action": "recovery_file_required" if run["state"] == "blocked" else None} + if run["state"] == "blocked": + next_action = "recovery_file_required" + elif run["state"] == "awaiting_host" and run["phase"] is None: + next_action = "claim_handoff" + elif run["state"] == "awaiting_host": + next_action = "handoff_submission_saved" + else: + next_action = None + return { + "run_id": run_id, + "state": run["state"], + "phase": run["phase"], + "version": run["version"], + "active_attempt": { + key: active.get(key) + for key in ("id", "status", "pid", "pgid", "heartbeat_at") + } if active else None, + "handoff": self._handoff_payload(handoff, include_packet=False) if handoff else None, + "next_action": next_action, + } finally: store.close() def events(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any]: @@ -227,11 +318,70 @@ def cancel(self, run_id: str) -> dict[str, Any]: elif run["state"] == "queued" and run["phase"] == "launching": version = store.cancel_launching(run_id) elif run["state"] == "queued" and run["phase"] is None: version = store.cancel_queued(run_id) elif run["state"] in {"running", "cancelling"}: version, _ = store.request_cancel(run_id) + elif run["state"] == "awaiting_host": version = store.cancel_host_wait(run_id) elif run["state"] in TERMINAL_STATES: version = run["version"] else: raise ConflictError("run requires recovery before cancellation") return {"run_id": run_id, "state": store.run(run_id)["state"], "version": version} finally: store.close() + def handoff_claim( + self, + run_id: str, + expected_version: int, + owner: str, + prior_claim: dict[str, Any] | None = None, + ) -> dict[str, Any]: + if type(expected_version) is not int or expected_version < 1: + raise ContractError("handoff expected version is invalid") + if not isinstance(owner, str) or not owner: + raise ContractError("handoff owner is required") + decoded = self._decode_claim(prior_claim) if prior_claim is not None else None + store = self._store() + try: + claim = store.claim_handoff(run_id, expected_version, owner, decoded) + snapshot = store.handoff_snapshot(run_id) + if snapshot is None: # Defensive: claim_handoff just verified it. + raise ConflictError("claimed handoff is missing") + return { + "run_id": run_id, + "state": "awaiting_host", + "phase": None, + "version": claim.run_version, + "action": claim.action, + "claim": self._claim_payload(claim), + "handoff": self._handoff_payload(snapshot, include_packet=True), + } + finally: + store.close() + + def handoff_complete( + self, + run_id: str, + claim: dict[str, Any], + decision: dict[str, Any], + ) -> dict[str, Any]: + decoded = self._decode_claim(claim) + if decoded.run_id != run_id: + raise ConflictError("handoff completion claim targets a different run") + store = self._store() + try: + submission = store.record_handoff_submission(run_id, decoded, decision) + run = store.run(run_id) + return { + "run_id": run_id, + "state": run["state"], + "phase": run["phase"], + "version": run["version"], + "handoff_id": submission.handoff_id, + "submission_id": submission.submission_id, + "submission_hash": submission.submission_hash, + "disposition": submission.disposition, + "recorded_run_version": submission.recorded_run_version, + "replayed": submission.replayed, + } + finally: + store.close() + def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[str, Any]: store = self._store() launch: tuple[int, Path, str] | None = None diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index c8e1e4a..24d3814 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -891,6 +891,43 @@ def _handoff_claim_from_row(row: sqlite3.Row, action: str) -> HandoffClaim: action=action, ) + @classmethod + def _handoff_snapshot_from_rows( + cls, + run_id: str, + run: sqlite3.Row, + handoff: sqlite3.Row | None, + claim_row: sqlite3.Row | None, + ) -> HandoffSnapshot | None: + if handoff is None: + return None + packet_json = handoff["packet_json"] + if hashlib.sha256(packet_json.encode()).hexdigest() != handoff["packet_sha256"]: + raise ConflictError("persisted handoff packet hash does not match") + try: + packet = json.loads(packet_json) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("persisted handoff packet is invalid") from exc + if not isinstance(packet, dict) or canonical_json(packet) != packet_json: + raise ConflictError("persisted handoff packet is not canonical finite JSON") + claim = cls._handoff_claim_from_row(claim_row, "current") if claim_row else None + return HandoffSnapshot( + run_id=run_id, + run_state=run["state"], + run_phase=run["phase"], + run_version=run["version"], + handoff_id=handoff["id"], + sequence=handoff["sequence"], + status=handoff["status"], + packet=packet, + packet_sha256=handoff["packet_sha256"], + created_run_version=handoff["created_run_version"], + submitted_run_version=handoff["submitted_run_version"], + created_at=handoff["created_at"], + closed_at=handoff["closed_at"], + claim=claim, + ) + def publish_handoff( self, run_id: str, @@ -988,7 +1025,7 @@ def handoff_snapshot(self, run_id: str) -> HandoffSnapshot | None: handoff = self.connection.execute( "SELECT * FROM handoffs WHERE run_id=? ORDER BY sequence DESC LIMIT 1", (run_id,), ).fetchone() - claim = None + claim_row = None if handoff: claim_row = self.connection.execute( "SELECT run_id,handoff_id,owner_id,fencing_token,lease_expires_at," @@ -996,29 +1033,8 @@ def handoff_snapshot(self, run_id: str) -> HandoffSnapshot | None: "AND handoff_id=? AND active=1", (run["version"], run_id, handoff["id"]), ).fetchone() - if claim_row: - claim = self._handoff_claim_from_row(claim_row, "current") - if not handoff: - self.connection.execute("COMMIT") - return None - packet_json = handoff["packet_json"] - if hashlib.sha256(packet_json.encode()).hexdigest() != handoff["packet_sha256"]: - raise ConflictError("persisted handoff packet hash does not match") - snapshot = HandoffSnapshot( - run_id=run_id, - run_state=run["state"], - run_phase=run["phase"], - run_version=run["version"], - handoff_id=handoff["id"], - sequence=handoff["sequence"], - status=handoff["status"], - packet=json.loads(packet_json), - packet_sha256=handoff["packet_sha256"], - created_run_version=handoff["created_run_version"], - submitted_run_version=handoff["submitted_run_version"], - created_at=handoff["created_at"], - closed_at=handoff["closed_at"], - claim=claim, + snapshot = self._handoff_snapshot_from_rows( + run_id, run, handoff, claim_row, ) self.connection.execute("COMMIT") return snapshot @@ -1070,7 +1086,8 @@ def claim_handoff( or prior_claim.handoff_id != row["handoff_id"] or prior_claim.owner_id != owner_id or row["owner_id"] != owner_id - or prior_claim.fencing_token != row["fencing_token"]): + or prior_claim.fencing_token != row["fencing_token"] + or prior_claim.expires_at != row["lease_expires_at"]): raise ConflictError("handoff renewal claim is stale or expired") token, action = row["fencing_token"], "renewed" self.connection.execute( @@ -1429,15 +1446,32 @@ def events_page(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str consumed = page[-1]["id"] if page else after return {"events": [{**dict(row), "payload": json.loads(row["payload"])} for row in page], "next_cursor": consumed, "has_more": more} - def status_snapshot(self, run_id: str) -> tuple[dict[str, Any], dict[str, Any] | None]: + def status_snapshot( + self, run_id: str, + ) -> tuple[dict[str, Any], dict[str, Any] | None, HandoffSnapshot | None]: self.connection.execute("BEGIN") try: run = self.connection.execute("SELECT * FROM runs WHERE id=?", (run_id,)).fetchone() if not run: raise ContractError("run does not exist") attempt = self.connection.execute("SELECT * FROM attempts WHERE run_id=? ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() + handoff = self.connection.execute( + "SELECT * FROM handoffs WHERE run_id=? ORDER BY sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + claim_row = None + if handoff: + claim_row = self.connection.execute( + "SELECT run_id,handoff_id,owner_id,fencing_token,lease_expires_at," + "? AS run_version FROM claims WHERE run_id=? AND kind='host' " + "AND handoff_id=? AND active=1", + (run["version"], run_id, handoff["id"]), + ).fetchone() + handoff_snapshot = self._handoff_snapshot_from_rows( + run_id, run, handoff, claim_row, + ) self.connection.execute("COMMIT") - return dict(run), dict(attempt) if attempt else None + return dict(run), dict(attempt) if attempt else None, handoff_snapshot except Exception: self.connection.execute("ROLLBACK") raise diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 5e480bd..78e1ad3 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -103,6 +103,67 @@ def test_status_events_result_cancel_and_resume_operations(self): self.assert_success_envelope(payload, response) service.resume.assert_called_once_with("run-1", {"attempt_id": "a-1", "disposition": "confirm_dead"}) + def test_handoff_claim_renew_and_complete_dispatch_parsed_objects(self): + claim_payload = { + "schema_version": 1, + "run_id": "run-1", + "handoff_id": "handoff-1", + "owner": "terminal-a", + "fencing_token": 7, + "expires_at": "2026-09-15T06:00:00+00:00", + "run_version": 11, + } + claim_file = self.root / "claim.json" + claim_file.write_text(json.dumps(claim_payload)) + decision_payload = { + "schema_version": 1, + "submission_id": "submission-1", + "submission_hash": "a" * 64, + "disposition": "accept", + "reason": "accepted", + "evidence_refs": [], + } + decision_file = self.root / "decision.json" + decision_file.write_text(json.dumps(decision_payload)) + + service = mock.Mock() + response = { + "run_id": "run-1", "state": "awaiting_host", "version": 12, + "claim": claim_payload, + } + service.handoff_claim.return_value = response + code, payload, stderr = self.invoke([ + "handoff", "claim", "run-1", "--expected-version", "11", + "--owner", "terminal-a", "--claim-file", str(claim_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.handoff_claim.assert_called_once_with( + "run-1", 11, "terminal-a", claim_payload, + ) + + service = mock.Mock() + response = { + "run_id": "run-1", "state": "awaiting_host", + "phase": "handoff_submitted", "replayed": False, + } + service.handoff_complete.return_value = response + code, payload, stderr = self.invoke([ + "handoff", "complete", "run-1", "--claim-file", str(claim_file), + "--decision-file", str(decision_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.handoff_complete.assert_called_once_with( + "run-1", claim_payload, decision_payload, + ) + + code, payload, _ = self.invoke(["handoff"]) + self.assertEqual(code, 64) + self.assertEqual(payload["error"]["code"], "INPUT_INVALID") + def test_parser_and_json_file_failures_are_input_errors(self): code, payload, _ = self.invoke(["status"]) self.assertEqual(code, 64) diff --git a/test/core/test_handoff_service.py b/test/core/test_handoff_service.py new file mode 100644 index 0000000..e6e3b7d --- /dev/null +++ b/test/core/test_handoff_service.py @@ -0,0 +1,209 @@ +from __future__ import annotations + +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.store import ConflictError, Store, request_hash + + +class HandoffServiceTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-handoff-service-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], + check=True, + ) + (self.repo / "README").write_text("base\n") + subprocess.run(["git", "-C", str(self.repo), "add", "README"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.service = Service(self.runtime) + + def waiting_run(self, key: str = "handoff") -> tuple[str, dict, int]: + packet = { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "instructions": "Review the frozen candidate.", + } + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + claim = store.claim_start(self.repo, key, {"task": key}, "preflight") + version = store.complete_preparation( + claim.run_id, + claim.fencing_token, + {"base_oid": "a" * 40, "target_oid": "b" * 40}, + package_path="/frozen/package", + package_digest="package-digest", + ) + reservation = store.reserve_attempt( + claim.run_id, version, "supervisor", "package-digest", + ) + version = store.mark_attempt_running( + reservation, 101, 101, "process-start-id", + ) + snapshot = store.publish_handoff( + claim.run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + packet, + ) + return claim.run_id, packet, snapshot.run_version + finally: + store.close() + + @staticmethod + def decision( + submission_id: str = "submission-1", + disposition: str = "accept", + reason: str = "accepted", + ) -> dict: + body = { + "schema_version": 1, + "submission_id": submission_id, + "disposition": disposition, + "reason": reason, + "evidence_refs": [], + } + return {**body, "submission_hash": request_hash(body)} + + def test_status_claim_complete_and_exact_replay_share_one_saved_run(self): + run_id, packet, version = self.waiting_run() + waiting = self.service.status(run_id) + self.assertEqual( + (waiting["state"], waiting["phase"], waiting["version"], waiting["next_action"]), + ("awaiting_host", None, version, "claim_handoff"), + ) + self.assertEqual(waiting["handoff"]["status"], "open") + self.assertNotIn("packet", waiting["handoff"]) + self.assertNotIn("fencing_token", waiting["handoff"]) + + acquired = self.service.handoff_claim(run_id, version, "terminal-a") + self.assertEqual(acquired["action"], "acquired") + self.assertEqual(acquired["handoff"]["packet"], packet) + self.assertEqual(acquired["handoff"]["claimed_by"], "terminal-a") + public_claim = acquired["claim"] + self.assertEqual(set(public_claim), { + "schema_version", "run_id", "handoff_id", "owner", + "fencing_token", "expires_at", "run_version", + }) + + renewed = self.service.handoff_claim( + run_id, acquired["version"], "terminal-a", public_claim, + ) + self.assertEqual(renewed["action"], "renewed") + self.assertEqual( + renewed["claim"]["fencing_token"], public_claim["fencing_token"], + ) + with self.assertRaises(ConflictError): + self.service.handoff_claim( + run_id, renewed["version"], "terminal-a", public_claim, + ) + + decision = self.decision() + completed = self.service.handoff_complete(run_id, renewed["claim"], decision) + self.assertEqual( + (completed["state"], completed["phase"], completed["disposition"]), + ("awaiting_host", "handoff_submitted", "accept"), + ) + self.assertFalse(completed["replayed"]) + replay = self.service.handoff_complete(run_id, renewed["claim"], decision) + self.assertTrue(replay["replayed"]) + self.assertEqual( + replay["recorded_run_version"], completed["recorded_run_version"], + ) + + def test_cancel_awaiting_host_is_terminal_and_keeps_replay_idempotent(self): + run_id, _, version = self.waiting_run("cancel-wait") + acquired = self.service.handoff_claim(run_id, version, "terminal-a") + decision = self.decision() + completed = self.service.handoff_complete(run_id, acquired["claim"], decision) + cancelled = self.service.cancel(run_id) + self.assertEqual(cancelled["state"], "cancelled") + result = self.service.result(run_id) + self.assertTrue(result["ready"]) + self.assertEqual(result["state"], "cancelled") + replay = self.service.handoff_complete(run_id, acquired["claim"], decision) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay["state"], "cancelled") + self.assertEqual( + replay["recorded_run_version"], completed["recorded_run_version"], + ) + + def test_public_claim_shape_and_run_binding_are_strict(self): + run_id, _, version = self.waiting_run("strict-claim") + with self.assertRaises(ContractError): + self.service.handoff_complete(run_id, {"schema_version": 1}, self.decision()) + acquired = self.service.handoff_claim(run_id, version, "terminal-a") + wrong_run = dict(acquired["claim"], run_id="different-run") + with self.assertRaises(ConflictError): + self.service.handoff_complete(run_id, wrong_run, self.decision()) + naive_expiry = dict(acquired["claim"], expires_at="2026-09-15T05:00:00") + with self.assertRaises(ContractError): + self.service.handoff_complete(run_id, naive_expiry, self.decision()) + + def test_independent_cli_processes_claim_and_complete_the_saved_handoff(self): + run_id, packet, version = self.waiting_run("cli-handoff") + environment = os.environ.copy() + environment["PYTHONPATH"] = str(ROOT / "plugin/core/src") + + def invoke(arguments): + result = subprocess.run( + [sys.executable, "-P", "-m", "devsquad.cli", *arguments], + cwd=self.root, + env=environment, + text=True, + capture_output=True, + check=False, + ) + self.assertEqual(result.stderr, "") + self.assertEqual(result.returncode, 0, result.stdout) + return json.loads(result.stdout) + + claimed = invoke([ + "handoff", "claim", run_id, + "--expected-version", str(version), + "--owner", "second-terminal", + "--runtime-dir", str(self.runtime), + "--json", + ]) + self.assertTrue(claimed["ok"]) + self.assertEqual(claimed["data"]["handoff"]["packet"], packet) + claim_file = self.root / "host-claim.json" + claim_file.write_text(json.dumps(claimed["data"]["claim"])) + decision = self.decision() + decision_file = self.root / "host-decision.json" + decision_file.write_text(json.dumps(decision)) + + completed = invoke([ + "handoff", "complete", run_id, + "--claim-file", str(claim_file), + "--decision-file", str(decision_file), + "--runtime-dir", str(self.runtime), + "--json", + ]) + self.assertTrue(completed["ok"]) + self.assertEqual(completed["data"]["phase"], "handoff_submitted") + self.assertEqual(completed["data"]["submission_hash"], decision["submission_hash"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 9e9ddfe..929bfe3 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -196,6 +196,14 @@ def test_claim_cas_renewal_and_expired_takeover(self): self.assertEqual(renewed.action, "renewed") self.assertEqual(renewed.fencing_token, first.fencing_token) self.assertGreater(renewed.expires_at, first.expires_at) + with self.assertRaises(ConflictError): + self.store.claim_handoff( + run_id, + renewed.run_version, + "host-a", + first, + now=first_time + timedelta(minutes=6), + ) with self.assertRaises(ConflictError): self.store.claim_handoff( run_id, diff --git a/test/core/test_m2_cross_process.py b/test/core/test_m2_cross_process.py index 950b8c6..a593e2d 100644 --- a/test/core/test_m2_cross_process.py +++ b/test/core/test_m2_cross_process.py @@ -13,7 +13,7 @@ sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) from devsquad.service import Service -from devsquad.store import Store +from devsquad.store import Store, request_hash def service_start(runtime, task, key, barrier, results): @@ -34,6 +34,28 @@ def service_cancel(runtime, run_id, barrier, results): results.put(("error", type(exc).__name__, str(exc))) +def service_handoff_claim(runtime, run_id, version, owner, barrier, results): + try: + barrier.wait(timeout=10) + value = Service(Path(runtime)).handoff_claim(run_id, version, owner) + results.put(( + "ok", value["claim"]["owner"], value["claim"]["fencing_token"], + )) + except Exception as exc: + results.put(("error", type(exc).__name__, str(exc))) + + +def service_handoff_complete(runtime, run_id, claim, decision, barrier, results): + try: + barrier.wait(timeout=10) + value = Service(Path(runtime)).handoff_complete(run_id, claim, decision) + results.put(( + "ok", value["replayed"], value["recorded_run_version"], + )) + except Exception as exc: + results.put(("error", type(exc).__name__, str(exc))) + + def reserve_writer(database, artifacts, run_id, version, owner, barrier, results): store = None try: @@ -90,6 +112,34 @@ def run_processes(self, targets): results.close() results.join_thread() + def waiting_handoff(self, key): + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + claim = store.claim_start(self.repo, key, {"task": key}, "preflight") + version = store.complete_preparation( + claim.run_id, + claim.fencing_token, + {"head": "fixed"}, + package_path="/frozen/package", + package_digest="package-digest", + ) + reservation = store.reserve_attempt( + claim.run_id, version, "supervisor", "package-digest", + ) + version = store.mark_attempt_running( + reservation, 101, 101, "process-start-id", + ) + handoff = store.publish_handoff( + claim.run_id, + version, + reservation.attempt_token, + reservation.supervisor_token, + {"schema_version": 1, "candidate_sha256": "c" * 64}, + ) + return claim.run_id, handoff.run_version + finally: + store.close() + def test_identical_public_starts_share_one_run_across_processes(self): args = (str(self.runtime), self.task, "same-process-key") outcomes = self.run_processes([(service_start,args),(service_start,args)]) @@ -150,6 +200,38 @@ def test_repeated_cancel_is_idempotent_across_processes(self): result = Service(self.runtime).result(claim.run_id) self.assertEqual([item["name"] for item in result["artifacts"]], ["result-receipt.json"]) + def test_two_processes_cannot_claim_one_host_handoff(self): + run_id, version = self.waiting_handoff("handoff-claim-race") + outcomes = self.run_processes([ + (service_handoff_claim, (str(self.runtime), run_id, version, "host-a")), + (service_handoff_claim, (str(self.runtime), run_id, version, "host-b")), + ]) + self.assertEqual(sorted(outcome[0] for outcome in outcomes), ["error", "ok"]) + self.assertEqual( + next(outcome for outcome in outcomes if outcome[0] == "error")[1], + "ConflictError", + ) + + def test_identical_handoff_completions_replay_across_processes(self): + run_id, version = self.waiting_handoff("handoff-complete-race") + acquired = Service(self.runtime).handoff_claim(run_id, version, "host-a") + body = { + "schema_version": 1, + "submission_id": "submission-1", + "disposition": "accept", + "reason": "accepted", + "evidence_refs": [], + } + decision = {**body, "submission_hash": request_hash(body)} + common = (str(self.runtime), run_id, acquired["claim"], decision) + outcomes = self.run_processes([ + (service_handoff_complete, common), + (service_handoff_complete, common), + ]) + self.assertTrue(all(outcome[0] == "ok" for outcome in outcomes), outcomes) + self.assertEqual(sorted(outcome[1] for outcome in outcomes), [False, True]) + self.assertEqual(len({outcome[2] for outcome in outcomes}), 1) + if __name__ == "__main__": unittest.main() From d02b8081efc6f82bb015b50741e91894cc303e05 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 15 Sep 2026 17:03:59 +0530 Subject: [PATCH 043/197] fix: reconcile gated runner crashes --- plugin/core/src/devsquad/service.py | 8 +- plugin/core/src/devsquad/store.py | 197 ++++++++++++++++++++++++- plugin/core/src/devsquad/supervisor.py | 131 +++++++++++++++- test/core/test_m2_cross_process.py | 69 +++++++++ test/core/test_m2_supervisor_gate.py | 28 ++++ test/core/test_service.py | 147 ++++++++++++++++++ 6 files changed, 570 insertions(+), 10 deletions(-) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 74b130e..2ff1f1f 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -319,6 +319,9 @@ def cancel(self, run_id: str) -> dict[str, Any]: elif run["state"] == "queued" and run["phase"] is None: version = store.cancel_queued(run_id) elif run["state"] in {"running", "cancelling"}: version, _ = store.request_cancel(run_id) elif run["state"] == "awaiting_host": version = store.cancel_host_wait(run_id) + elif run["state"] == "blocked": + from .supervisor import Supervisor + version = Supervisor(store).cancel_orphan(run_id) elif run["state"] in TERMINAL_STATES: version = run["version"] else: raise ConflictError("run requires recovery before cancellation") return {"run_id": run_id, "state": store.run(run_id)["state"], "version": version} @@ -393,7 +396,10 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st from .supervisor import Supervisor attempt=store.attempt(run_id) disposition = Supervisor(store).import_durable(run_id) if attempt and attempt.get("exit_record") else Supervisor(store).recover(run_id) - return {"run_id": run_id, "disposition": disposition, "launched": False} + if disposition != "requeued": + return {"run_id": run_id, "disposition": disposition, "launched": False} + run = store.run(run_id) + version = run["version"] if run["state"] == "queued" and run["phase"] == "preparing": submitted = json.loads(run["submitted_request"]) claim = store.reclaim_preparation( diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 24d3814..d20379c 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -308,6 +308,7 @@ def _terminal_receipt( phase: str, payload: Any, *, + attempt_id: str | None = None, now: str | None = None, ) -> tuple[Path, str, int, str]: finished_at = now or _utc_now() @@ -316,7 +317,7 @@ def _terminal_receipt( "run_id": run_id, "state": terminal_state, "phase": phase, - "attempt_id": None, + "attempt_id": attempt_id, "returncode": None, "cancelled": terminal_state == "cancelled", "timed_out": False, @@ -811,6 +812,200 @@ def block_recovery(self, run_id: str, attempt_token: str, reason: str, *, releas self.connection.execute("ROLLBACK") raise + def recover_unstarted_attempt( + self, run_id: str, attempt_token: str, reason: str, + ) -> tuple[int, str]: + """Recover a dead gated runner that never published a child identity. + + The supervisor must first establish that the runner is dead and the + atomic child record is absent. The inner gate cannot open before that + record is durable, so no worker command can have executed in this case. + """ + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,child_record FROM attempts " + "WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if (not run or not attempt or run["state"] not in {"running", "cancelling"} + or run["phase"] is not None + or attempt["status"] not in {"running", "cancelling"}): + raise ConflictError("unstarted attempt recovery is fenced") + if not attempt["child_record"]: + raise ConflictError("unstarted attempt has no durable child-record path") + if Path(attempt["child_record"]).exists() or Path(attempt["child_record"]).is_symlink(): + raise ConflictError("unstarted attempt now has a child identity record") + now = _utc_now() + if run["state"] == "cancelling": + path, digest, size, receipt_time = self._terminal_receipt( + run_id, + "cancelled", + "cancelling", + None, + attempt_id=attempt["id"], + now=now, + ) + version = self._reference_terminal_receipt( + run_id, run["version"], path, digest, size, receipt_time, + ) + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "reason": reason, + "receipt": "result-receipt.json", + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, payload, now), + ) + disposition = "cancelled" + else: + version = run["version"] + 1 + self.connection.execute( + "UPDATE attempts SET status='recovery_required',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='queued',phase=NULL,version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.unstarted_attempt_recovered',?,?)", + (run_id, version, canonical_json({ + "attempt_id": attempt["id"], "reason": reason, + }), now), + ) + disposition = "requeued" + self.connection.execute("COMMIT") + return version, disposition + except Exception: + self.connection.execute("ROLLBACK") + raise + + def request_recovery_cancel(self, run_id: str, attempt_token: str) -> int: + """Persist cancellation intent for an ownerless blocked attempt.""" + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("recovery cancellation is fenced") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"] + if run["state"] == "cancelling" and attempt["status"] == "cancelling": + self.connection.execute("COMMIT") + return run["version"] + if (run["state"] != "blocked" or run["phase"] != "recovery_required" + or attempt["status"] not in {"ownership_ambiguous", "recovery_required"}): + raise ConflictError("run is not awaiting recovery cancellation") + version, now = run["version"] + 1, _utc_now() + self.connection.execute( + "UPDATE attempts SET status='cancelling',heartbeat_at=?,finished_at=NULL WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE runs SET state='cancelling',phase='recovery_cleanup'," + "version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({"attempt_id": attempt["id"]}) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelling',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + + def finish_recovery_cancel( + self, run_id: str, attempt_token: str, reason: str, + ) -> int: + """Finish an ownerless cancellation only after absence is confirmed.""" + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("recovery cancellation completion is fenced") + if run["state"] in TERMINAL_STATES: + self.connection.execute("COMMIT") + return run["version"] + if (run["state"] != "cancelling" or run["phase"] != "recovery_cleanup" + or attempt["status"] != "cancelling"): + raise ConflictError("recovery cancellation is not active") + now = _utc_now() + path, digest, size, receipt_time = self._terminal_receipt( + run_id, + "cancelled", + "recovery_cleanup", + None, + attempt_id=attempt["id"], + now=now, + ) + version = self._reference_terminal_receipt( + run_id, run["version"], path, digest, size, receipt_time, + ) + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "reason": reason, + "receipt": "result-receipt.json", + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + def fail_launch(self, reservation: AttemptReservation, reason: str) -> int: self.connection.execute("BEGIN IMMEDIATE") try: diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 32ca8f0..23541a2 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -96,6 +96,20 @@ def _live_group_exists(pgid: int) -> bool: return False +def _read_child_identity(path: Path) -> tuple[int, int, str]: + if path.is_symlink() or not path.is_file(): + raise ValueError("child identity record is not a regular file") + value = json.loads(path.read_text()) + if not isinstance(value, dict) or set(value) != {"pid", "pgid", "process_start_id"}: + raise ValueError("child identity record fields are invalid") + if (type(value["pid"]) is not int or value["pid"] <= 0 + or type(value["pgid"]) is not int or value["pgid"] <= 0 + or not isinstance(value["process_start_id"], str) + or not value["process_start_id"]): + raise ValueError("child identity record values are invalid") + return value["pid"], value["pgid"], value["process_start_id"] + + class BoundedDrain: def __init__(self, stream: BinaryIO, limit: int): self.stream, self.limit = stream, limit @@ -201,6 +215,8 @@ def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, o gate_read, gate_write = os.pipe() process = None stdin_stream = None + identity_committed = False + gate_released = False try: if spec.stdin_path is not None: stdin_stream = _open_stdin_artifact(spec.stdin_path) @@ -222,7 +238,10 @@ def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, o started=process_start_identity(process.pid) if started is None: raise RuntimeError("gated child has no strong process identity") self.store.mark_attempt_running(reservation,process.pid,process.pid,started,paths) - os.write(gate_write,b"1"); os.close(gate_write) + identity_committed = True + self._release_runner_gate(gate_write) + gate_released = True + os.close(gate_write) return DurableAttempt(reservation,process,paths) except Exception: if stdin_stream is not None: @@ -236,9 +255,24 @@ def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, o try: os.killpg(process.pid,signal.SIGKILL) except ProcessLookupError: pass process.wait(timeout=2) - self.store.fail_launch(reservation,"durable gated launch failed") + if not identity_committed: + self.store.fail_launch(reservation,"durable gated launch failed") + elif not gate_released: + self.store.recover_unstarted_attempt( + run_id, reservation.attempt_token, + "runner gate could not be released", + ) + else: + self.store.block_recovery( + run_id, reservation.attempt_token, + "coordinator failed after releasing the runner gate", + ) raise + @staticmethod + def _release_runner_gate(gate_write: int) -> None: + os.write(gate_write, b"1") + def wait_durable(self, handle: DurableAttempt, timeout_seconds: float) -> int: deadline=time.monotonic()+timeout_seconds + self.grace_seconds + 5 while handle.process.poll() is None and time.monotonic() str: return "ownership_ambiguous" if not receipt_path.is_file(): child_path=Path(attempt["child_record"] or "") - if child_path.is_file(): + if child_path.exists() or child_path.is_symlink(): try: - child=json.loads(child_path.read_text()) - if inspect_process(child["pid"],child["pgid"],child["process_start_id"])!="dead": + child = _read_child_identity(child_path) + if inspect_process(*child)!="dead": self.store.block_recovery(run_id,attempt["attempt_token"],"runner died while its child may still be live") return "ownership_ambiguous" - except (OSError,ValueError,KeyError,TypeError): + except (OSError, ValueError, TypeError, json.JSONDecodeError): self.store.block_recovery(run_id,attempt["attempt_token"],"child identity record is invalid") return "ownership_ambiguous" - self.store.block_recovery(run_id,attempt["attempt_token"],"runner died without an exit receipt",release_writer=True) - return "recovery_required" + self.store.block_recovery(run_id,attempt["attempt_token"],"runner and child died without an exit receipt",release_writer=True) + return "recovery_required" + try: + _, disposition = self.store.recover_unstarted_attempt( + run_id, + attempt["attempt_token"], + "runner died before publishing the gated child identity", + ) + return disposition + except ConflictError: + current = self.store.attempt(run_id) + run = self.store.run(run_id) + if (current and current["attempt_token"] == attempt["attempt_token"] + and current["status"] == "recovery_required" + and run["state"] == "queued"): + return "requeued" + if run["state"] in {"succeeded", "failed", "cancelled"}: + return run["state"] + raise try: receipt=json.loads(receipt_path.read_text()) if (type(receipt.get("returncode")) is not int @@ -317,6 +368,70 @@ def _terminate_durable(self, handle: DurableAttempt) -> None: while handle.process.poll() is None and time.monotonic() int: + """Cancel a blocked worker whose durable runner is confirmed dead.""" + attempt = self.store.attempt(run_id) + if (not attempt or attempt["status"] not in { + "ownership_ambiguous", "recovery_required", "cancelling", + }): + raise ConflictError("run has no orphaned attempt to cancel") + runner_identity = ( + attempt["pid"], attempt["pgid"], attempt["process_start_id"], + ) + if all(value is None for value in runner_identity): + runner = "dead" + elif (type(runner_identity[0]) is int and runner_identity[0] > 0 + and type(runner_identity[1]) is int and runner_identity[1] > 0 + and isinstance(runner_identity[2], str) and runner_identity[2]): + runner = inspect_process(*runner_identity) + else: + raise ConflictError("orphan runner identity record is invalid") + if runner != "dead": + raise ConflictError("orphan runner identity is not confirmed dead") + + child = None + classification = "dead" + child_record = attempt["child_record"] + if child_record: + child_path = Path(child_record) + else: + child_path = None + if child_path is not None and (child_path.exists() or child_path.is_symlink()): + try: + child = _read_child_identity(child_path) + except (OSError, ValueError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError("orphan child identity record is invalid") from exc + classification = inspect_process(*child) + if classification == "ambiguous": + raise ConflictError("orphan child identity is ambiguous or reused") + + self.store.request_recovery_cancel(run_id, attempt["attempt_token"]) + if child is not None and classification == "live": + classification = inspect_process(*child) + if classification == "ambiguous": + raise ConflictError("orphan child identity changed before cancellation") + if classification == "live": + try: + os.killpg(child[1], signal.SIGTERM) + except ProcessLookupError: + pass + deadline = time.monotonic() + self.grace_seconds + while _live_group_exists(child[1]) and time.monotonic() < deadline: + time.sleep(0.05) + if _live_group_exists(child[1]): + try: + os.killpg(child[1], signal.SIGKILL) + except ProcessLookupError: + pass + deadline = time.monotonic() + max(self.grace_seconds, 2.0) + while _live_group_exists(child[1]) and time.monotonic() < deadline: + time.sleep(0.05) + if _live_group_exists(child[1]): + raise ConflictError("orphan child survived bounded cancellation") + return self.store.finish_recovery_cancel( + run_id, attempt["attempt_token"], "orphaned worker cleanup confirmed", + ) + def _persist_output(self, handle: RunningAttempt) -> dict[str, Any]: stdout_meta, stderr_meta = handle.stdout.finish(), handle.stderr.finish() run_id, token = handle.reservation.run_id, handle.reservation.attempt_token diff --git a/test/core/test_m2_cross_process.py b/test/core/test_m2_cross_process.py index a593e2d..5f5dde5 100644 --- a/test/core/test_m2_cross_process.py +++ b/test/core/test_m2_cross_process.py @@ -3,10 +3,13 @@ import json import multiprocessing +import os from pathlib import Path +import signal import subprocess import sys import tempfile +import time import unittest ROOT = Path(__file__).resolve().parents[2] @@ -14,6 +17,7 @@ from devsquad.service import Service from devsquad.store import Store, request_hash +from devsquad.supervisor import inspect_process def service_start(runtime, task, key, barrier, results): @@ -140,6 +144,15 @@ def waiting_handoff(self, key): finally: store.close() + def wait_state(self, run_id, expected, timeout=8): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = Service(self.runtime).status(run_id) + if status["state"] in expected: + return status + time.sleep(0.05) + self.fail(f"run did not reach {expected}: {Service(self.runtime).status(run_id)}") + def test_identical_public_starts_share_one_run_across_processes(self): args = (str(self.runtime), self.task, "same-process-key") outcomes = self.run_processes([(service_start,args),(service_start,args)]) @@ -200,6 +213,62 @@ def test_repeated_cancel_is_idempotent_across_processes(self): result = Service(self.runtime).result(claim.run_id) self.assertEqual([item["name"] for item in result["artifacts"]], ["result-receipt.json"]) + def test_running_cancel_race_reaps_runner_and_worker_once(self): + service = Service(self.runtime) + started = service.start( + self.task, "running-cancel-race", _internal_fake_delay=30, + ) + self.wait_state(started["run_id"], {"running"}) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + child = None + try: + attempt = store.attempt(started["run_id"]) + child_path = Path(attempt["child_record"]) + deadline = time.monotonic() + 5 + while not child_path.is_file() and time.monotonic() < deadline: + time.sleep(0.02) + self.assertTrue(child_path.is_file()) + child = json.loads(child_path.read_text()) + finally: + store.close() + try: + args = (str(self.runtime), started["run_id"]) + outcomes = self.run_processes([(service_cancel, args), (service_cancel, args)]) + self.assertTrue(all(outcome[0] == "ok" for outcome in outcomes), outcomes) + self.assertTrue(all(outcome[1] in {"cancelling", "cancelled"} for outcome in outcomes)) + self.wait_state(started["run_id"], {"cancelled"}) + + deadline = time.monotonic() + 5 + while (inspect_process( + attempt["pid"], attempt["pgid"], attempt["process_start_id"], + ) != "dead" and time.monotonic() < deadline): + time.sleep(0.02) + self.assertEqual( + inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]), + "dead", + ) + self.assertEqual( + inspect_process(child["pid"], child["pgid"], child["process_start_id"]), + "dead", + ) + result = service.result(started["run_id"]) + self.assertTrue(result["ready"]) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + types = [ + event["type"] + for event in store.events_page(started["run_id"], limit=1000)["events"] + ] + self.assertEqual(types.count("run.cancelling"), 1) + self.assertEqual(types.count("run.cancelled"), 1) + finally: + store.close() + finally: + if child and inspect_process( + child["pid"], child["pgid"], child["process_start_id"], + ) == "live": + os.killpg(child["pgid"], signal.SIGKILL) + def test_two_processes_cannot_claim_one_host_handoff(self): run_id, version = self.waiting_handoff("handoff-claim-race") outcomes = self.run_processes([ diff --git a/test/core/test_m2_supervisor_gate.py b/test/core/test_m2_supervisor_gate.py index ec1bfec..d304458 100644 --- a/test/core/test_m2_supervisor_gate.py +++ b/test/core/test_m2_supervisor_gate.py @@ -81,6 +81,30 @@ def test_spawn_failure_is_fenced_and_releases_worktree_writer(self): reservation = self.store.reserve_attempt(second, second_version, "other-owner", "package") self.assertTrue(reservation.attempt_token) + def test_durable_outer_gate_failure_requeues_without_running_command(self): + run_id, version = self._ready_run("outer-gate-failure") + marker = self.root / "GATE_COMMAND_EXECUTED" + code = f"from pathlib import Path;Path({str(marker)!r}).write_text('executed')" + source = str(ROOT / "plugin" / "core" / "src") + with mock.patch.dict(os.environ, {"PYTHONPATH": source}), \ + mock.patch.object( + self.supervisor, + "_release_runner_gate", + side_effect=OSError("synthetic gate failure"), + ): + with self.assertRaisesRegex(OSError, "synthetic gate failure"): + self.supervisor.launch_durable( + run_id, + version, + self._spec(sys.executable, "-c", code), + "owner", + "package", + ) + run = self.store.run(run_id) + self.assertEqual((run["state"], run["phase"]), ("queued", None)) + self.assertEqual(self.store.attempt(run_id)["status"], "recovery_required") + self.assertFalse(marker.exists()) + def test_output_is_bounded_but_full_stream_is_accounted(self): run_id, version = self._ready_run("output") code = "import sys;sys.stdout.write('o'*1000);sys.stderr.write('e'*2000)" @@ -167,6 +191,10 @@ def test_durable_launch_passes_an_opened_regular_stdin_artifact(self): linked_run,linked_version,linked_spec,"owner","package", ) self.assertEqual(self.store.run(linked_run)["state"],"blocked") + version = self.supervisor.cancel_orphan(linked_run) + self.assertEqual(self.store.run(linked_run)["state"], "cancelled") + self.assertEqual(self.store.run(linked_run)["version"], version) + self.assertIsNotNone(self.store.artifact_named(linked_run, "result-receipt.json")) def test_recovery_never_signals_an_ambiguous_identity(self): run_id, version = self._ready_run("ambiguous") diff --git a/test/core/test_service.py b/test/core/test_service.py index 347621e..1656880 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -13,6 +13,7 @@ ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.contracts import ExecutionIdentity, LaunchSpec from devsquad.service import Service from devsquad.store import ConflictError, Store from devsquad.supervisor import Supervisor, inspect_process @@ -37,6 +38,29 @@ def crash_after_attempt_reservation(database, artifacts, run_id, expected_versio os._exit(23) +def crash_after_runner_identity( + database, artifacts, run_id, expected_version, package_path, + package_digest, repo, marker, +): + os.environ["PYTHONPATH"] = package_path + store = Store(Path(database), Path(artifacts)) + identity = ExecutionIdentity("fixture", "1", None, None, None, None) + command = ( + sys.executable, + "-c", + f"from pathlib import Path; Path({marker!r}).write_text('executed')", + ) + spec = LaunchSpec( + 1, "fixture", "cli_exec", command, repo, None, 30, identity, + ) + supervisor = Supervisor(store) + supervisor._release_runner_gate = lambda _: os._exit(24) + supervisor.launch_durable( + run_id, expected_version, spec, "crashed-after-identity", package_digest, + ) + os._exit(99) + + class ServiceTest(unittest.TestCase): def setUp(self): self.temp = tempfile.TemporaryDirectory(prefix="devsquad-service-") @@ -193,6 +217,70 @@ def test_process_crash_after_attempt_reservation_is_recoverable(self): finally: store.close() + def test_crash_after_runner_identity_before_gate_requeues_without_execution(self): + with mock.patch.object(self.service,"_spawn_daemon",return_value=0): + started=self.service.start( + self.task,"identity-before-gate",_internal_fake_delay=.01, + ) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + run=store.run(started["run_id"]) + expected_version=run["version"] + package_path=run["package_path"] + package_digest=run["package_digest"] + finally: + store.close() + marker=self.root/"PRE_GATE_COMMAND_EXECUTED" + context=multiprocessing.get_context("spawn") + process=context.Process( + target=crash_after_runner_identity, + args=( + str(self.runtime/"state.sqlite3"),str(self.runtime/"artifacts"), + started["run_id"],expected_version,package_path,package_digest, + str(self.repo),str(marker), + ), + ) + process.start(); process.join(timeout=10) + if process.is_alive(): + process.terminate(); process.join(timeout=2) + self.assertEqual(process.exitcode,24) + + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + attempt=store.attempt(started["run_id"]) + deadline=time.monotonic()+5 + while (inspect_process( + attempt["pid"],attempt["pgid"],attempt["process_start_id"], + )=="live" and time.monotonic() Date: Tue, 15 Sep 2026 17:06:24 +0530 Subject: [PATCH 044/197] test: prove crash-safe artifact import --- test/core/test_service.py | 68 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 68 insertions(+) diff --git a/test/core/test_service.py b/test/core/test_service.py index 1656880..2778592 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -61,6 +61,13 @@ def crash_after_runner_identity( os._exit(99) +def crash_after_artifact_finalize_before_import(database, artifacts, run_id): + store = Store(Path(database), Path(artifacts)) + store.commit_durable_import = lambda *args, **kwargs: os._exit(25) + Supervisor(store).import_durable(run_id) + os._exit(99) + + class ServiceTest(unittest.TestCase): def setUp(self): self.temp = tempfile.TemporaryDirectory(prefix="devsquad-service-") @@ -511,6 +518,67 @@ def test_coordinator_crash_imports_runner_receipt_once(self): try: self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM attempts WHERE run_id=?",(started["run_id"],)).fetchone()[0],1) finally: store.close() + def test_importer_crash_after_artifact_finalize_has_no_false_completion(self): + started=self.service.start( + self.task,"artifact-import-crash",_internal_fake_delay=.2, + ) + self.wait_state(started["run_id"],{"running"}) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + attempt=store.attempt(started["run_id"]) + owner=store.connection.execute( + "SELECT owner_id FROM supervisor_claims WHERE run_id=?", + (started["run_id"],), + ).fetchone()[0] + os.kill(int(owner.split(":",1)[1]),signal.SIGKILL) + receipt=Path(attempt["exit_record"]) + deadline=time.monotonic()+5 + while not receipt.is_file() and time.monotonic() Date: Wed, 16 Sep 2026 00:40:38 +0530 Subject: [PATCH 045/197] fix: close final M2 recovery races --- plugin/core/src/devsquad/service.py | 3 + plugin/core/src/devsquad/store.py | 15 ++-- test/core/test_handoff_store.py | 129 ++++++++++++++++++++++++++++ test/core/test_service.py | 91 ++++++++++++++++++++ 4 files changed, 230 insertions(+), 8 deletions(-) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 2ff1f1f..fe1a35e 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -317,6 +317,9 @@ def cancel(self, run_id: str) -> dict[str, Any]: if run["state"] == "queued" and run["phase"] == "preparing": version = store.cancel_preparing(run_id) elif run["state"] == "queued" and run["phase"] == "launching": version = store.cancel_launching(run_id) elif run["state"] == "queued" and run["phase"] is None: version = store.cancel_queued(run_id) + elif run["state"] == "cancelling" and run["phase"] == "recovery_cleanup": + from .supervisor import Supervisor + version = Supervisor(store).cancel_orphan(run_id) elif run["state"] in {"running", "cancelling"}: version, _ = store.request_cancel(run_id) elif run["state"] == "awaiting_host": version = store.cancel_host_wait(run_id) elif run["state"] == "blocked": diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index d20379c..6529e3a 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1147,9 +1147,9 @@ def publish_handoff( raise ContractError("handoff publication requires valid ownership and packet fields") packet_json = canonical_json(packet) packet_sha256 = hashlib.sha256(packet_json.encode()).hexdigest() - timestamp = _authoritative_now(now).isoformat() self.connection.execute("BEGIN IMMEDIATE") try: + timestamp = _authoritative_now(now).isoformat() row = self.connection.execute( "SELECT r.state,r.phase,r.version,a.id AS attempt_id,a.status AS attempt_status," "s.fencing_token AS supervisor_token,s.active AS supervisor_active " @@ -1251,11 +1251,11 @@ def claim_handoff( raise ContractError("handoff claim requires a run version and owner") if prior_claim is not None and not isinstance(prior_claim, HandoffClaim): raise ContractError("prior handoff claim is invalid") - current = _authoritative_now(now) - timestamp = current.isoformat() - expires_at = (current + timedelta(seconds=HOST_LEASE_SECONDS)).isoformat() self.connection.execute("BEGIN IMMEDIATE") try: + current = _authoritative_now(now) + timestamp = current.isoformat() + expires_at = (current + timedelta(seconds=HOST_LEASE_SECONDS)).isoformat() row = self.connection.execute( "SELECT r.state,r.phase,r.version,h.id AS handoff_id,h.status," "c.kind,c.owner_id,c.fencing_token,c.active,c.handoff_id AS claim_handoff_id," @@ -1389,12 +1389,12 @@ def record_handoff_submission( ) decision_json = canonical_json(decision) evidence_json = canonical_json(evidence_refs) - current = _authoritative_now(now) - timestamp = current.isoformat() rejection = None result = None self.connection.execute("BEGIN IMMEDIATE") try: + current = _authoritative_now(now) + timestamp = current.isoformat() run = self.connection.execute( "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), ).fetchone() @@ -1538,10 +1538,9 @@ def record_handoff_submission( return result def cancel_host_wait(self, run_id: str, *, now: datetime | None = None) -> int: - current = _authoritative_now(now) - timestamp = current.isoformat() self.connection.execute("BEGIN IMMEDIATE") try: + timestamp = _authoritative_now(now).isoformat() run = self.connection.execute( "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), ).fetchone() diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 929bfe3..11cde12 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -8,7 +8,9 @@ import sys import tempfile import threading +import time import unittest +from unittest import mock ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "plugin/core/src")) @@ -90,6 +92,88 @@ def decision(submission_id="submission-1", disposition="accept", reason="accepte } return {**body, "submission_hash": request_hash(body)} + def claim_with_deadline(self, key, deadline): + run_id, _, snapshot = self.waiting_run(key) + claim = self.store.claim_handoff( + run_id, + snapshot.run_version, + "host", + now=deadline - timedelta(minutes=1), + ) + expires_at = deadline.isoformat() + self.store.connection.execute( + "UPDATE claims SET lease_expires_at=? WHERE run_id=?", + (expires_at, run_id), + ) + return run_id, HandoffClaim( + run_id=claim.run_id, + handoff_id=claim.handoff_id, + owner_id=claim.owner_id, + fencing_token=claim.fencing_token, + expires_at=expires_at, + run_version=claim.run_version, + action=claim.action, + ) + + def run_after_independent_writer_wait(self, deadline, operation): + """Run an operation whose BEGIN IMMEDIATE is blocked across a lease expiry.""" + ready = threading.Event() + begin_attempted = threading.Event() + release_started = threading.Event() + failures = [] + connection = self.store.connection + + class BeginNotifyingConnection: + def execute(self, statement, *args, **kwargs): + if statement == "BEGIN IMMEDIATE": + begin_attempted.set() + return connection.execute(statement, *args, **kwargs) + + def __getattr__(self, name): + return getattr(connection, name) + + def hold_lock(): + writer = sqlite3.connect(self.database, timeout=5, isolation_level=None) + try: + writer.execute("PRAGMA busy_timeout=5000") + writer.execute("BEGIN IMMEDIATE") + ready.set() + if not begin_attempted.wait(5): + raise AssertionError("handoff operation did not attempt its write transaction") + # Give the caller time to enter SQLite's busy wait before releasing the lock. + time.sleep(0.05) + release_started.set() + writer.execute("COMMIT") + except BaseException as exc: # Preserve thread failures for the assertion owner. + failures.append(exc) + ready.set() + finally: + if writer.in_transaction: + writer.execute("ROLLBACK") + writer.close() + + thread = threading.Thread(target=hold_lock) + thread.start() + self.store.connection = BeginNotifyingConnection() + try: + self.assertTrue(ready.wait(5), "independent SQLite writer did not acquire its lock") + self.assertEqual(failures, []) + before_expiry = deadline - timedelta(seconds=1) + after_expiry = deadline + timedelta(seconds=1) + with mock.patch( + "devsquad.store._authoritative_now", + side_effect=lambda value=None: ( + after_expiry if release_started.is_set() else before_expiry + ), + ): + return operation() + finally: + begin_attempted.set() + thread.join(5) + self.store.connection = connection + self.assertFalse(thread.is_alive()) + self.assertEqual(failures, []) + def test_schema_four_fixture_migrates_to_host_handoffs(self): old_database = self.root / "schema-four.sqlite3" connection = sqlite3.connect(old_database) @@ -252,6 +336,51 @@ def claim(owner): self.assertEqual(len(claims), 1) self.assertEqual(len(conflicts), 1) + def test_renewal_checks_expiry_after_waiting_for_write_lock(self): + deadline = datetime(2026, 9, 15, 5, 2, tzinfo=timezone.utc) + run_id, claim = self.claim_with_deadline("renewal-lock", deadline) + version_before = self.store.run(run_id)["version"] + with self.assertRaisesRegex(ConflictError, "stale or expired"): + self.run_after_independent_writer_wait( + deadline, + lambda: self.store.claim_handoff( + run_id, + version_before, + claim.owner_id, + claim, + ), + ) + self.assertEqual(self.store.run(run_id)["version"], version_before) + persisted = self.store.connection.execute( + "SELECT lease_expires_at,renewed_at FROM claims WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(tuple(persisted), (claim.expires_at, None)) + + def test_completion_checks_expiry_after_waiting_for_write_lock(self): + deadline = datetime(2026, 9, 15, 5, 2, tzinfo=timezone.utc) + run_id, claim = self.claim_with_deadline("completion-lock", deadline) + version_before = self.store.run(run_id)["version"] + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.run_after_independent_writer_wait( + deadline, + lambda: self.store.record_handoff_submission( + run_id, claim, self.decision(), + ), + ) + run = self.store.run(run_id) + self.assertEqual( + (run["state"], run["phase"], run["version"]), + ("awaiting_host", None, version_before + 1), + ) + rejection = self.store.connection.execute( + "SELECT outcome,rejection_code,recorded_run_version " + "FROM handoff_submissions WHERE handoff_id=?", + (claim.handoff_id,), + ).fetchone() + self.assertEqual( + tuple(rejection), ("rejected", "expired_claim", version_before + 1), + ) + def test_expired_submission_is_audited_and_cannot_displace_takeover(self): run_id, _, snapshot = self.waiting_run() started = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) diff --git a/test/core/test_service.py b/test/core/test_service.py index 2778592..2238c4d 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -68,6 +68,19 @@ def crash_after_artifact_finalize_before_import(database, artifacts, run_id): os._exit(99) +def crash_after_recovery_cancel_commits(database, artifacts, run_id): + store = Store(Path(database), Path(artifacts)) + request_recovery_cancel = store.request_recovery_cancel + + def request_then_crash(cancel_run_id, attempt_token): + request_recovery_cancel(cancel_run_id, attempt_token) + os._exit(26) + + store.request_recovery_cancel = request_then_crash + Supervisor(store).cancel_orphan(run_id) + os._exit(99) + + class ServiceTest(unittest.TestCase): def setUp(self): self.temp = tempfile.TemporaryDirectory(prefix="devsquad-service-") @@ -362,6 +375,84 @@ def test_runner_death_with_live_child_can_be_safely_cancelled(self): )=="live": os.killpg(child["pgid"],signal.SIGKILL) + def test_public_cancel_resumes_crashed_orphan_cleanup_once(self): + started=self.service.start( + self.task,"orphan-cancel-restart",_internal_fake_delay=30, + ) + self.wait_state(started["run_id"],{"running"}) + child=None + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + attempt=store.attempt(started["run_id"]) + child_path=Path(attempt["child_record"]) + deadline=time.monotonic()+5 + while not child_path.is_file() and time.monotonic() Date: Wed, 16 Sep 2026 00:44:42 +0530 Subject: [PATCH 046/197] docs: accept durable M2 milestone --- docs/plans/engineering-team/M2-STATUS.md | 88 +++++++------ docs/plans/engineering-team/RESUME.md | 120 ++++++++---------- docs/plans/engineering-team/backlog.json | 13 +- .../evidence/M2-durable-runs-2026-09-16.json | 47 +++++++ 4 files changed, 163 insertions(+), 105 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/M2-durable-runs-2026-09-16.json diff --git a/docs/plans/engineering-team/M2-STATUS.md b/docs/plans/engineering-team/M2-STATUS.md index 9bf21bf..0296094 100644 --- a/docs/plans/engineering-team/M2-STATUS.md +++ b/docs/plans/engineering-team/M2-STATUS.md @@ -1,40 +1,56 @@ # M2 implementation status -M2 is **in progress**. The store checkpoint is `0c2929c`; the bounded -supervisor lifecycle checkpoints are `0ee4cc0` and `df955f4`; the first -service checkpoint is `a0794a9`. +M2 is **complete** at implementation checkpoint `ddb6f51`. The final gate +passes 121 core tests with `ResourceWarning` promoted to an error and all 10 +Bash regression files/202 assertions. No provider invocation was used for the +M2 gate. -| Store requirement | Evidence | Status | +| Requirement | Evidence | Status | |---|---|---| -| SQLite WAL and packaged schema migration | Fresh install opens the packaged migration; four independent processes concurrently initialize one database | verified offline | -| Canonical request idempotency before mutable snapshot resolution | Concurrent connections return one run for the same key/body; a changed body conflicts | verified offline | -| Canonical project identity | Main and linked Git worktrees resolve to the same absolute common directory and project ID | verified offline | -| Preparing-owner fencing, recovery and cancellation | Runs remain `queued` with a private `preparing` phase; explicit recovery rotates the fencing token and replays the immutable submitted request; competing reclaimers yield one owner; stale completion after cancel/reclaim conflicts | verified offline | -| Transactional projections and events | Compare-and-swap run version and append-only event commit together under concurrent writers | verified offline | -| Atomic hash-verified artifacts | Content-addressed files finalize before reference; references increment run version with an event; duplicates and terminal mutation cannot clobber prior content | verified offline | -| Schema migration and future refusal | Source fixtures upgrade from versions 1 and 3; an installed wheel applies migration 004 from a schema-3 fixture; newer unsupported versions are rejected | verified offline | -| Supervisor and writer fencing | Transactional claims allow one supervisor and one active writer per worktree; ambiguous ownership retains the database fence | verified offline | -| Strong process identity and recovery | Darwin start second+microsecond identity is stable; live children remain owned without relaunch; dead and reused identities receive distinct recovery dispositions and reused IDs are never signalled | verified offline | -| Bounded process lifecycle | Direct argv runs in a new session; heartbeat, PID/PGID/start identity, token and package digest persist; stdout/stderr drain continuously with truncation and full-stream hashes | verified offline | -| Timeout and cancellation | Intent precedes verified TERM/KILL; TERM-resistant root and descendant disappear before completion; durable timeout remains failed when TERM exits 0; every queued/preparing/launching cancellation atomically publishes a receipt | partial: cross-process cancel/cleanup races remain | -| Durable gated launch | A persisted attempt runner and an inner worker gate prevent task execution before strong runner and child identity records; the runner owns timeout, cancel polling, bounded spool files, exact opened stdin and an fsynced exit receipt | verified offline | -| Coordinator-loss recovery | Preparing work is reclaimed under a rotated token; a real process crash after attempt reservation is fenced and requeued; a separate coordinator killed while the runner lives is not relaunched; two processes import one completed receipt atomically | partial: post-identity/pre-gate and orphan-child recovery remain | -| Frozen package | Every regular runtime asset is hashed, copied atomically, fsynced, stored on the run and verified before initial launch or resume; `-P` regressions prevent repository shadowing of internal modules | verified offline | -| Public workflow guard | Public branch-review tasks end with `CAPABILITY_UNAVAILABLE` until M3; only the private test argument can invoke the M2 fake step | verified offline | -| Result and event reads | Status is a transactional run/attempt snapshot; cursors always report the last consumed position and `has_more`; all terminal paths require a hash-checked result receipt | verified offline | -| Predecessor link | Public and fixture paths validate terminal same-project predecessors before mutable filesystem work; valid lineage survives later preparation failure while missing, active and foreign-project predecessors fail with a receipt | verified offline | -| CLI envelope and wait behavior | Exact v1 envelopes/exit codes, M2 operation dispatch, observation-only Ctrl-C, and terminal `start --wait` mappings | verified offline | - -Checkpoint `f9f2ffa` passes 94 core tests and 202 shell assertions. The core -suite includes an installed-wheel schema-3-to-4 migration, two-process receipt -import and preparation-claim races, real crashes after reservation, -repository-retarget/shadow resistance, exact durable stdin, and a timeout whose -TERM handler exits zero. Detached supervisors are reaped without the earlier -`ResourceWarning`s. An independent review found project-identity and lineage -defects in the first preparation-recovery patch; both have regressions and were -fixed before this checkpoint. - -Remaining acceptance work is post-identity/pre-gate and orphan-child -reconciliation, cross-process cancel/start/writer races, host-handoff storage -and claim fencing, and a final independent M2 gate. This is not a claim of a -working engineering workflow; branch-review execution begins in M3. +| SQLite ledger, WAL and packaged migrations | Concurrent fresh-database initialization; source fixtures migrate from schemas 1, 3 and 4; a wheel installed outside the checkout applies all packaged migrations through schema 5 and rejects a future schema | verified offline | +| Canonical start idempotency and project identity | Independent processes return one run for identical requests and conflict for a changed body; linked Git worktrees share the canonical common-directory project ID | verified offline | +| Preparing-owner recovery | Interrupted preflight replays the immutable submitted request under a rotated fence; competing reclaimers yield one owner; stale owners and repository retargeting are rejected | verified offline | +| Transactional state, events and artifacts | Projection/version compare-and-swap and events share a transaction; artifacts are atomically finalized and hash checked before reference | verified offline | +| Supervisor and writer fencing | Database claims permit one active worktree writer; strong process identity distinguishes live, dead and ambiguous/reused processes | verified offline | +| Durable gated launch | The persisted runner cannot open the inner worker gate until runner identity and spool paths are committed; a crash after identity but before gate release requeues without executing the command | verified offline | +| Coordinator-loss recovery | A launching shell can exit while another process observes/imports the run; a live runner is not relaunched; completed receipts import exactly once across processes | verified offline | +| Orphan-child recovery | A dead runner with a live recorded child blocks relaunch; owned recovery cancellation signals only the verified child group and confirms absence before terminal state | verified offline | +| Interrupted recovery cancellation | A spawned canceller exits immediately after persisting `recovery_cleanup`; the next public cancel resumes cleanup, emits one terminal event/receipt and repeated cancel is harmless | verified offline | +| Timeout, cancellation and stdin | Exact regular-file stdin bytes reach the child; a timeout stays failed even when TERM exits zero; cancellation intent precedes bounded TERM/KILL and terminal state follows confirmed cleanup | verified offline | +| Cross-process races | Independent-process tests cover identical/conflicting starts, worktree writer claims, queued/running cancel, receipt import, host claims and handoff completion replay | verified offline | +| Host handoffs | Schema-5 packets, bounded claims, renewal/takeover, submission replay, late audit, awaiting-host cancellation and CLI/service operations are fenced and durable | verified offline | +| Lease authorization under contention | Real independent SQLite writer locks force renewal and completion to wait across expiry; authoritative time is sampled only after the write transaction is acquired | verified offline | +| CLI and durable reads | Exact v1 envelopes/exit mappings, `start --wait`, observation-only Ctrl-C, transactional status, stable cursors and hash-verified terminal results | verified offline | +| Public workflow boundary | Public branch review fails explicitly with `CAPABILITY_UNAVAILABLE`; only the private fake-step hook exercises M2 lifecycle | verified offline | + +## Acceptance mapping + +- Concurrent identical starts converge on one run; a different request under + the same key conflicts. +- Event/projection mutations and terminal artifact references remain + transactional under independent writers. +- Killing a coordinator or runner never creates a second writer. Ambiguous or + reused process identities are not signalled. +- A crash after artifact finalization but before its database transaction + creates no false completion; resume imports the same finalized bytes once. +- Cancellation survives caller loss and never records terminal cancellation + before the verified child group is absent. +- Schema migration is exercised from the preceding schema through an installed + wheel, and newer unsupported schemas are refused. +- A second process can inspect and operate on the same saved run and can claim + a host handoff; stale, expired and lock-delayed completions cannot advance it. + +## Independent review closure + +The final bounded review found two P1 races and no additional defect in its +targeted scope: lease time was sampled before SQLite lock acquisition, and an +interruption after orphan-cancel intent could leave public cancel stuck. Both +were fixed at `ddb6f51` and have deterministic contention/process-crash +regressions in the 121-test gate. + +## Boundary + +M2 proves the durable execution substrate, not a working engineering workflow. +`branch-review` intentionally remains unavailable until M3 implements frozen +review workspaces, deterministic profile routing, reviewer/check/lead flow and +bound reports. M3 is the first user-usable product checkpoint. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index bfcae23..b7310d7 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,62 +2,41 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 15, 2026 +## Current position — September 16, 2026 - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. -- Build branch: `codex/engineering-team`. `main` is the published runtime baseline. -- Accepted M1 implementation checkpoint: `97a10f0`. -- Probe cleanup hardening checkpoint: `7b5c41c`. -- Accepted M1 status/evidence checkpoint: `c50c6b4`. -- Native correlation/probe checkpoint: `97a10f0`. The saved integrated probe - passed; its private receipt SHA256 is - `324c2ce6154936ecf71a8bf913a195befd4251efc6dbb8bbab3dc0dbe1b86df8`. - Exact private receipt directory: +- Build branch: `codex/engineering-team`. `main` remains the published runtime + baseline. Inspect current refs before acting; later build checkpoints may be + local and must not be discarded. +- M1 is accepted for bounded truthful invocation/preparation at `97a10f0`. + Its saved integrated Codex receipt SHA256 is + `324c2ce6154936ecf71a8bf913a195befd4251efc6dbb8bbab3dc0dbe1b86df8`; + the private receipt remains outside the repository under `/Users/Dikshant/.devsquad/private-probes/native-codex-20260907T025419Z-97a10f0ae1e8`. -- M1 is accepted for its bounded invocation/preparation scope. The next - milestone is M2. -- M2 store foundation checkpoint: `0c2929c`. M2 remains in progress; its - supervisor, service operations and recovery slices are not accepted yet. -- Supervisor lifecycle checkpoint: `0ee4cc0`. It completes migration 002, - fenced supervisor/writer ownership, strong process identity, bounded output, - heartbeat, timeout/cancel cleanup and conservative recovery. M2 remains in - progress because durable service operations and a detached supervisor entry - point are pending. -- Cleanup-race hardening checkpoint: `df955f4`. Cleanup inventory fails closed, - timeout/cancel identity races retain ownership fencing, and 62 core tests pass. -- Durable service checkpoint `cefd193` is explicit WIP and gated-runner - checkpoint `a0794a9` passed 67 core tests plus 202 shell assertions. The - preserved recovery candidate is `f76df41`. Adversarial hardening checkpoint - `5aa2e74` and CLI/wheel checkpoint `f0d29a7` pass 82 core tests plus all 202 - shell assertions. Preparation/receipt checkpoint `144b0b2`, reviewed - project-identity fix `c6a7a23`, and gated-launch/stdin checkpoint `f9f2ffa` - now pass 94 core tests plus all 202 shell assertions. The current implementation - checkpoint replaces the unsafe anonymous-pipe launch with a persisted, - gated attempt runner and migration 003 durable paths. The runner cannot - launch the internal fixture worker until its strong identity and spool paths - are committed; it owns output drains, heartbeat, cancel polling and an - atomic exit record. Ordered migration discovery, transactional status, - stable event cursors and fenced preparation failure are implemented. - Recovery/import of a runner exit record is transactionally idempotent under - two-process races; timeouts cannot become success after a clean TERM exit; - repository-local Python packages cannot shadow the frozen internal runner. - Exact CLI envelopes, `start --wait`, and an installed-wheel schema-3-to-4 - migration gate are covered. Interrupted preparation rotates its fencing token - and replays the canonical submitted request without permitting repository - retargeting. Every public pre-attempt terminal path publishes a durable - receipt; public/fixture predecessor validation covers valid, missing, active - and foreign-project links, including unrelated failure and repository-loss - cases. A real crash after gated-launch reservation is recoverable, durable - stdin preserves exact bytes, and detached supervisor handles are reaped. - Independent review found and drove the project-identity/lineage regressions. - Remaining M2 work includes post-identity/pre-gate and orphan-child recovery, - cross-process cancellation/start/writer races, host-handoff claim fencing, - and a final post-fix M2 gate. Public branch-review execution is explicitly - failed as `CAPABILITY_UNAVAILABLE` until M3; only the internal test hook can - run the fake step. M2 remains in progress. -- GitHub build branch contains the cleanup/architecture checkpoint `55e93a2`; later implementation checkpoints are local. Inspect the actual current refs before acting. -- User wants **Sol to implement, with Astra reviewing**, and explicitly wants work preserved across Plus-plan usage interruptions. -- Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M2 is in progress. +- M2 is accepted at `ddb6f51`. Its durable store, gated runner, cancellation, + recovery, schema-5 host handoffs and CLI/service operations pass 121 core + tests with `ResourceWarning` promoted to failure and 202 Bash assertions. + See [M2-STATUS.md](M2-STATUS.md) and the + [portable redacted evidence](evidence/M2-durable-runs-2026-09-16.json). +- Final independent M2 review found two P1 races and no other defect in its + bounded target. `ddb6f51` fixes both: lease authorization now samples time + after acquiring the SQLite write transaction, and public cancel resumes an + interrupted `recovery_cleanup`. Both have deterministic regressions. +- The next milestone is M3. Public `branch-review` still fails explicitly as + `CAPABILITY_UNAVAILABLE`; M2 is infrastructure, not yet a usable engineering + workflow. Do not advertise DevSquad as ready for real review until M3 passes. +- Current provider readiness is external to M2: Claude CLI is not logged in; + Grok CLI authentication expired; Gemini CLI's individual-account path is + unsupported and its supported successor is Antigravity; Antigravity is + authenticated but headless execution still lacks scoped permission/trust. + Do not retry these blocked paths, buy credits, use paid API fallback or + change global provider settings. Resume a provider gate only after the user + completes the corresponding normal login/permission action. +- User wants the implementation orchestrated efficiently and preserved across + Plus-plan interruptions. Avoid recursive subagent fan-out: it consumed the + shared window rapidly without advancing M3. +- Full assignment remains **M1–M7 plus C1**, as specified in + [SOL-HANDOFF.md](SOL-HANDOFF.md). M3 is next. ## Completed and preserved @@ -69,11 +48,12 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 34 tests passed | +| Python core discovery | 121 tests passed at the M2 acceptance gate | | Bash 3.2 regression suite | 10 test files, 202 assertions passed | -| Wheel installation | Temporary venv resolves packaged schemas, adapters and shared taxonomy | +| Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 5 | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | | Integrated native adapter proof | Passed at `97a10f0`; gpt-5.5/low, read-only, correlated completion and confirmed process-group cleanup | +| M2 crash/race matrix | Real subprocess interruptions plus independent-process start, writer, cancel, import and host-handoff races passed at `ddb6f51` | The first two saved-probe invocations failed before `Popen` because of probe-only path/field defects, so neither launched Codex nor consumed a model @@ -84,21 +64,27 @@ The successful run retained separate stderr files of 138,030 and 285,644 bytes, supporting the diagnosis that an undrained stderr pipe caused the earlier apparent nonresponses. -The authoritative requirement matrix is [M1-STATUS.md](M1-STATUS.md); detailed evidence is [M1-invocation-core-2026-09-06.json](evidence/M1-invocation-core-2026-09-06.json). [backlog.json](backlog.json) marks M1 complete and M2 next. Grok workspace-write and unprobed Antigravity/Grok settings are not advertised as verified. +The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md) and +[M2-STATUS.md](M2-STATUS.md). [backlog.json](backlog.json) marks both complete +and M3 next. Unauthenticated, unsupported or permission-blocked provider paths +are not advertised as verified. ## Exact next work -1. Check Git state; preserve any new changes before doing further work. Read this file, M1-STATUS and the full Sol handoff. Do not restart the architecture exercise or reset to `main`. -2. Read the M2 section of [IMPLEMENTATION.md](IMPLEMENTATION.md), - [M2-STATUS.md](M2-STATUS.md), and the corresponding contracts before - editing. Continue with migration 005 host-handoff storage and claim fencing, - then post-identity/pre-gate and orphan-child recovery plus the remaining - cross-process cancellation/start/writer crash matrix. - Do not redo the accepted store or supervisor foundations or the fixes in - `5aa2e74`/`f0d29a7`. -3. Preserve M1 limitations and process-ownership boundaries. M1 acceptance is - not a claim that the full product exists, and no additional native probe is - needed for the accepted gate. +1. Check Git status and recent commits, preserving work newer than this note. + Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. +2. Read the M3 section of [IMPLEMENTATION.md](IMPLEMENTATION.md), the task, + Profile/Policy and workflow sections of [CONTRACTS.md](CONTRACTS.md), and + the selection amendment. Implement the earliest M3 dependency: strict + profile/policy loading plus deterministic frozen role bindings and fallback + decisions, with tests. Then add frozen review workspace preparation before + any real reviewer launch. +3. Preserve the M2 process/fencing boundaries. Extend the durable runner with + real workflow steps; do not bypass it with an in-process or ad-hoc provider + call. Keep public review unavailable until reviewer, declared checks, lead + disposition and bound receipt artifacts form a valid end-to-end slice. +4. Use offline fake adapters for development. Provider login/permission work is + a later live gate and must not block independent M3 implementation. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 398719c..57c8c14 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -11,7 +11,7 @@ "execution_brief": "SOL-HANDOFF.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", - "next_milestone": "M2", + "next_milestone": "M3", "milestones": [ { "id": "M1", @@ -36,7 +36,7 @@ "id": "M2", "title": "Durable runs, cancellation and recovery", "depends_on": ["M1"], - "status": "in_progress", + "status": "complete", "acceptance_section": "M2 — Persist jobs and own their processes", "evidence": [ { @@ -65,6 +65,15 @@ "artifact": "M2-STATUS.md", "recorded_at": "2026-09-15T11:06:00+05:30", "availability": "tracked tests" + }, + { + "kind": "milestone_acceptance", + "revision": "ddb6f51", + "command_or_action": "121 core tests with ResourceWarning promoted to error, 202 shell assertions, spawned-process crash/cancel recovery, independent-process races and installed-wheel schema-4-to-5 migration", + "outcome": "All M2 durability, cancellation, recovery and host-handoff gates pass; both final independent-review P1 findings are fixed with regressions; real branch review remains the explicit M3 boundary", + "artifact": "evidence/M2-durable-runs-2026-09-16.json", + "recorded_at": "2026-09-16T00:41:03+05:30", + "availability": "portable_redacted" } ], "blocker": null diff --git a/docs/plans/engineering-team/evidence/M2-durable-runs-2026-09-16.json b/docs/plans/engineering-team/evidence/M2-durable-runs-2026-09-16.json new file mode 100644 index 0000000..a1065e3 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M2-durable-runs-2026-09-16.json @@ -0,0 +1,47 @@ +{ + "schema_version": 1, + "milestone": "M2", + "status": "complete", + "implementation_revision": "ddb6f51", + "recorded_at": "2026-09-16T00:41:03+05:30", + "provider_invocations": 0, + "evidence": [ + { + "kind": "offline_test", + "command_or_action": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning python3 -m unittest discover -s test/core -v", + "outcome": "121 tests passed, including real subprocess crash recovery, independent-process races, deterministic SQLite lock contention, exact stdin, bounded cleanup and installed-wheel migration through schema 5", + "availability": "tracked tests" + }, + { + "kind": "offline_test", + "command_or_action": "bash test/run.sh", + "outcome": "10 Bash test files passed; 202 assertions preserved the Bash 3.2, optional-jq, wrapper and legacy contracts", + "availability": "tracked tests" + }, + { + "kind": "crash_and_race_matrix", + "command_or_action": "Spawned-process tests interrupt preparation, attempt reservation, post-runner-identity/pre-gate launch, live-runner coordination, orphan-child cancellation, exit import and artifact finalization before the database commit; independent processes race starts, writers, cancellation, claims and completion", + "outcome": "No duplicate run/writer/cleanup, no execution before the gate, no blind relaunch or ambiguous signal, no false completion and idempotent recovery", + "availability": "tracked tests" + }, + { + "kind": "host_handoff", + "command_or_action": "Service, CLI and Store tests publish, claim, renew, take over, complete, replay, audit late submissions and cancel schema-5 host handoffs", + "outcome": "Run-version and fencing-token ownership is durable; expiry is evaluated after acquiring the SQLite write transaction, including a real lock wait crossing the deadline", + "availability": "tracked tests" + }, + { + "kind": "package_test", + "command_or_action": "Build an offline wheel, install it into a fresh temporary environment outside the checkout with PYTHONPATH removed, and migrate an independent schema-4 fixture", + "outcome": "Packaged migrations through 005 are present, schema 4 upgrades to 5, and future schema versions are refused", + "availability": "reproducible locally" + } + ], + "independent_review": { + "status": "completed_with_findings_resolved", + "scope": "M2 crash recovery, host-handoff fencing and final artifact-import additions", + "outcome": "The reviewer found two P1 races and no other defect in the bounded target. Checkpoint ddb6f51 moves lease authorization after lock acquisition and makes recovery_cleanup resumable through public cancel; both findings have failing-boundary regressions in the final gate." + }, + "milestone_boundary": "Real branch-review workflow execution begins in M3; public M2 requests return CAPABILITY_UNAVAILABLE.", + "residual_blockers": [] +} From ca559906e1117be78f35f5fd5bdc299b2e1517a5 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 16 Sep 2026 00:53:30 +0530 Subject: [PATCH 047/197] feat: add deterministic M3 profile routing --- docs/plans/engineering-team/CONTRACTS.md | 19 + .../plans/engineering-team/examples/README.md | 9 +- plugin/core/pyproject.toml | 2 +- plugin/core/schemas/policy.schema.json | 7 +- plugin/core/schemas/profiles.schema.json | 31 ++ plugin/core/src/devsquad/contracts.py | 8 + plugin/core/src/devsquad/router.py | 390 ++++++++++++++++++ plugin/core/src/devsquad/validation.py | 46 +++ test/core/test_router.py | 360 ++++++++++++++++ 9 files changed, 868 insertions(+), 4 deletions(-) create mode 100644 plugin/core/schemas/profiles.schema.json create mode 100644 plugin/core/src/devsquad/router.py create mode 100644 test/core/test_router.py diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index 639fd90..35d29c2 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -68,6 +68,15 @@ required_tools[]; permission_policy; account_pool_id; billing_mode; quality_status; evidence_refs[] ``` +The path named by `routing.profiles_file` contains one strict profile registry, +not a bare array: `{schema_version:1, profiles:[Profile...], bindings:{...}}`. +Profile IDs are unique. Each binding is +`alias -> {profile_id, version}` and must target a profile in the same registry. +The registry's exact byte hash, every selected concrete profile/hash and every +resolved alias/version are frozen in preflight. Editing either file after that +cannot move an active run. Runtime-qualified binding promotion in M6 uses the +same versioned binding shape and affects new runs only. + `effort.transport` is `native`, `model_variant`, or `provider_default`. Values are native to that harness/model; never translate “high” into a numeric equivalent across vendors. Unsupported explicit effort fails validation. Provider default may be allowed, but its effective value remains unknown unless reported. `quality_status` is `unvalidated`, `trial`, `proven` or `suspended`, scoped to task class by policy evidence. Trial profiles are eligible only in explicitly permitted classes. Exact family and model IDs are mandatory for cross-model independence claims; inability to verify identity blocks that claim. Install-time discovery reports supported values and evidence (`documented`, `probed`, `unavailable`, `unknown`) with `checked_at`, CLI version and toolset hash. Selecting a known catalog entry verifies that it exists, not that the invocation used it: attempts retain separate `requested` and `observed` fields. Manually verified mappings may establish identity for a versioned harness; silent model fallback must never be labelled confirmed. @@ -78,6 +87,16 @@ Default selection is automatic among policy-eligible profiles. The host supplies `routing.overrides` defaults to `{}`. Each key is a model role supported by the chosen workflow, with value `{profile_id, fallback}`; `fallback` defaults to `none` and optionally allows `policy`. A pin constrains the initial profile; `none` also prohibits substitution/escalation to another profile. A pinned profile must pass the ordinary identity, capability, quality, permission and billing filters. Invalid pins fail validation; temporarily unavailable pins block. `policy` permits the normal qualified fallback list within the task budget. Roles without pins remain automatic. Record overrides and their origin in the routing receipt and separate them from automatic decisions in evaluations. +Each `policy.account_pools` entry has +`allowed_billing_modes[]`, positive `max_concurrency`, and optional +`unknown_capacity_policy` (`allow_bounded` by default or `block`). Runtime +availability is a separate snapshot with status `available`, `exhausted` or +`unknown`, a non-negative local `in_flight` count and an optional sourced +timestamp. Unknown capacity never becomes an invented allowance: +`allow_bounded` permits at most one unresolved in-flight job in that pool, +while `block` permits none. Declaring `paid_api` in a profile is insufficient; +the pool policy must explicitly allow that billing mode. + The [selection and Council amendment](SELECTION-AND-COUNCIL.md) gives examples and ownership boundaries. Evidence gathering/proposals are automatic. Initially promotions are reviewed; the [model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) permits tested model-binding promotion under an explicitly enabled, previously reviewed `guarded_auto` policy. Changing permissions, billing authority or the promotion policy itself remains reviewed. Optional C1 is outside the initial two-workflow schema until that gated extension ships. ## 3. Adapter execution boundary diff --git a/docs/plans/engineering-team/examples/README.md b/docs/plans/engineering-team/examples/README.md index b048d9e..a716462 100644 --- a/docs/plans/engineering-team/examples/README.md +++ b/docs/plans/engineering-team/examples/README.md @@ -5,6 +5,13 @@ These task files show the v1 shape. They are **not runnable against today's DevS - [branch-review.json](branch-review.json): a host lead receives the review packet; a failing report-only check becomes a finding. - [issue-delivery.json](issue-delivery.json): a headless lead resolves a bounded implementation/review workflow; required tests must pass. -The fixture test harness must install a `devsquad/profiles.json` and `devsquad/policy.json` inside its temporary repository, with two fictional model families and fake executables. Runtime configuration uses exact locally verified models and effort settings. These examples intentionally avoid embedding today's provider model names or pretending that fixture profiles are proven. +The fixture test harness must install a `devsquad/profiles.json` and +`devsquad/policy.json` inside its temporary repository, with two fictional +model families and fake executables. The profiles file is a strict +`{schema_version, profiles, bindings}` registry; the policy declares each +account pool's allowed billing modes and local concurrency limit. Runtime +configuration uses exact locally verified models and effort settings. These +examples intentionally avoid embedding today's provider model names or +pretending that fixture profiles are proven. The CLI and MCP service must validate equivalent task objects against the same generated schema. Validate paths/refs against the temporary repository only in integration tests; ordinary schema tests validate shape without touching the filesystem. diff --git a/plugin/core/pyproject.toml b/plugin/core/pyproject.toml index 5cec46b..63b4eff 100644 --- a/plugin/core/pyproject.toml +++ b/plugin/core/pyproject.toml @@ -25,5 +25,5 @@ where = ["src"] "share/devsquad/adapters/antigravity" = ["adapters/antigravity/adapter.json"] "share/devsquad/adapters/grok" = ["adapters/grok/adapter.json"] "share/devsquad/adapters" = ["adapters/classification-policy.conf"] -"share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/task.schema.json"] +"share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/profiles.schema.json", "schemas/task.schema.json"] "share/devsquad/profiles" = ["profiles/templates.json"] diff --git a/plugin/core/schemas/policy.schema.json b/plugin/core/schemas/policy.schema.json index aaba821..c5f8555 100644 --- a/plugin/core/schemas/policy.schema.json +++ b/plugin/core/schemas/policy.schema.json @@ -6,7 +6,10 @@ "schema_version":{"const":1},"id":{"type":"string","minLength":1},"version":{"type":"integer","minimum":1}, "roles":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"type":"array","minItems":1,"items":{"$ref":"#/$defs/candidate"}}}, "task_classes":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"enum":["unvalidated","trial","proven","suspended"]}},"require_different_model_for_review":{"type":"boolean"},"prefer_different_harness_for_review":{"type":"boolean"}, - "account_pools":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"type":"object"}},"experiment_budget":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"type":"integer","minimum":0}} + "account_pools":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"$ref":"#/$defs/account_pool"}},"experiment_budget":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"type":"integer","minimum":0}} }, - "$defs":{"candidate":{"type":"object","additionalProperties":false,"required":["kind","id"],"properties":{"kind":{"enum":["profile","alias"]},"id":{"type":"string","minLength":1}}}} + "$defs":{ + "candidate":{"type":"object","additionalProperties":false,"required":["kind","id"],"properties":{"kind":{"enum":["profile","alias"]},"id":{"type":"string","minLength":1}}}, + "account_pool":{"type":"object","additionalProperties":false,"required":["allowed_billing_modes","max_concurrency"],"properties":{"allowed_billing_modes":{"type":"array","minItems":1,"uniqueItems":true,"items":{"enum":["subscription","paid_api"]}},"max_concurrency":{"type":"integer","minimum":1},"unknown_capacity_policy":{"enum":["allow_bounded","block"]}}} + } } diff --git a/plugin/core/schemas/profiles.schema.json b/plugin/core/schemas/profiles.schema.json new file mode 100644 index 0000000..30ce652 --- /dev/null +++ b/plugin/core/schemas/profiles.schema.json @@ -0,0 +1,31 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/profiles-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "profiles", "bindings"], + "properties": { + "schema_version": {"const": 1}, + "profiles": { + "type": "array", + "minItems": 1, + "items": {"$ref": "profile.schema.json"} + }, + "bindings": { + "type": "object", + "propertyNames": {"minLength": 1}, + "additionalProperties": {"$ref": "#/$defs/binding"} + } + }, + "$defs": { + "binding": { + "type": "object", + "additionalProperties": false, + "required": ["profile_id", "version"], + "properties": { + "profile_id": {"type": "string", "minLength": 1}, + "version": {"type": "integer", "minimum": 1} + } + } + } +} diff --git a/plugin/core/src/devsquad/contracts.py b/plugin/core/src/devsquad/contracts.py index fcfece4..fbb132e 100644 --- a/plugin/core/src/devsquad/contracts.py +++ b/plugin/core/src/devsquad/contracts.py @@ -21,6 +21,14 @@ class ProfileUnsupported(ContractError): code = "PROFILE_UNSUPPORTED" +class CapabilityUnavailable(ContractError): + code = "CAPABILITY_UNAVAILABLE" + + +class PolicyDenied(ContractError): + code = "POLICY_DENIED" + + @dataclass(frozen=True) class ExecutionIdentity: harness: str diff --git a/plugin/core/src/devsquad/router.py b/plugin/core/src/devsquad/router.py new file mode 100644 index 0000000..5ed13f8 --- /dev/null +++ b/plugin/core/src/devsquad/router.py @@ -0,0 +1,390 @@ +"""Deterministic, side-effect-free M3 execution-profile selection.""" + +from __future__ import annotations + +from datetime import datetime +import hashlib +import json +from typing import Any + +from .contracts import ( + CapabilityUnavailable, + ContractError, + PolicyDenied, + ProfileUnsupported, +) +from .store import canonical_json +from .validation import validate_policy, validate_profile_registry, validate_task + + +QUALITY_RANK = {"unvalidated": 0, "trial": 1, "proven": 2} +ROLE_PERMISSIONS = { + "implementer": "workspace_write", + "reviewer": "read_only", + "lead": "read_only", + "researcher": "read_only", +} +WORKFLOW_ROLES = { + "branch-review": ("reviewer",), + "issue-delivery": ("implementer", "reviewer"), +} +CAPACITY_STATES = {"available", "exhausted", "unknown"} + + +def _strict_json(payload: bytes | str, label: str) -> tuple[dict[str, Any], str]: + if isinstance(payload, str): + encoded = payload.encode() + elif isinstance(payload, bytes): + encoded = payload + else: + raise ContractError(f"{label} must be JSON bytes or text") + + def object_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result = {} + for key, value in pairs: + if key in result: + raise ContractError(f"{label} contains duplicate key: {key}") + result[key] = value + return result + + def reject_constant(value: str) -> None: + raise ContractError(f"{label} contains non-finite number: {value}") + + try: + value = json.loads( + encoded.decode("utf-8"), + object_pairs_hook=object_pairs, + parse_constant=reject_constant, + ) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError(f"{label} is not valid UTF-8 JSON") from exc + if not isinstance(value, dict): + raise ContractError(f"{label} must contain a JSON object") + return value, hashlib.sha256(encoded).hexdigest() + + +def _availability_snapshot( + policy: dict[str, Any], availability: dict[str, Any] | None, +) -> dict[str, dict[str, Any]]: + supplied = {} if availability is None else availability + if not isinstance(supplied, dict): + raise ContractError("capacity availability must be an object") + unknown_pools = set(supplied) - set(policy["account_pools"]) + if unknown_pools: + raise ContractError(f"capacity references unknown account pools: {sorted(unknown_pools)}") + result = {} + for pool_id, pool_policy in policy["account_pools"].items(): + observation = supplied.get(pool_id, {"status": "unknown", "in_flight": 0}) + if not isinstance(observation, dict): + raise ContractError("capacity observation must be an object") + unknown = set(observation) - {"status", "in_flight", "observed_at"} + missing = {"status", "in_flight"} - set(observation) + if unknown or missing: + raise ContractError( + f"capacity observation fields invalid: unknown={sorted(unknown)} " + f"missing={sorted(missing)}" + ) + status, in_flight = observation["status"], observation["in_flight"] + if not isinstance(status, str) or status not in CAPACITY_STATES: + raise ContractError("capacity status is invalid") + if type(in_flight) is not int or in_flight < 0: + raise ContractError("capacity in_flight must be a non-negative integer") + observed_at = observation.get("observed_at") + if observed_at is not None: + if not isinstance(observed_at, str) or not observed_at: + raise ContractError("capacity observed_at must be a timestamp or null") + try: + parsed = datetime.fromisoformat(observed_at) + except ValueError as exc: + raise ContractError("capacity observed_at must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError("capacity observed_at must include a timezone") + result[pool_id] = { + "status": status, + "in_flight": in_flight, + "observed_at": observed_at, + "max_concurrency": pool_policy["max_concurrency"], + "unknown_capacity_policy": pool_policy.get( + "unknown_capacity_policy", "allow_bounded", + ), + } + return result + + +def _resolve_reference( + reference: dict[str, str], + profiles: dict[str, dict[str, Any]], + bindings: dict[str, dict[str, Any]], +) -> tuple[dict[str, Any] | None, dict[str, Any] | None, str | None]: + if reference["kind"] == "profile": + profile = profiles.get(reference["id"]) + return profile, None, None if profile else "profile_not_found" + binding = bindings.get(reference["id"]) + if binding is None: + return None, None, "alias_unbound" + profile = profiles.get(binding["profile_id"]) + if profile is None: # Defensive: registry validation already establishes this. + return None, None, "binding_target_missing" + return profile, { + "alias": reference["id"], + "version": binding["version"], + "profile_id": binding["profile_id"], + }, None + + +def _static_reason( + profile: dict[str, Any], + role: str, + minimum_quality: str, + pool_policy: dict[str, Any] | None, + implementer: dict[str, Any] | None, + require_different_model: bool, +) -> str | None: + if profile["quality_status"] == "suspended": + return "profile_suspended" + minimum_rank = QUALITY_RANK.get(minimum_quality) + profile_rank = QUALITY_RANK.get(profile["quality_status"]) + if minimum_rank is None or profile_rank is None or profile_rank < minimum_rank: + return "quality_below_task_minimum" + if profile["permission_policy"] != ROLE_PERMISSIONS[role]: + return "permission_mismatch" + if pool_policy is None: + return "account_pool_not_declared" + if profile["billing_mode"] not in pool_policy["allowed_billing_modes"]: + return "billing_mode_not_allowed" + if (role == "reviewer" and implementer is not None and require_different_model + and profile["model_family"] == implementer["model_family"] + and profile["model_id"] == implementer["model_id"]): + return "review_model_not_independent" + return None + + +def _capacity_reason( + profile: dict[str, Any], capacity: dict[str, dict[str, Any]], +) -> str | None: + pool = capacity[profile["account_pool_id"]] + if pool["status"] == "exhausted": + return "account_pool_exhausted" + if pool["in_flight"] >= pool["max_concurrency"]: + return "account_pool_concurrency_full" + if pool["status"] == "unknown": + if pool["unknown_capacity_policy"] == "block": + return "unknown_capacity_blocked" + if pool["in_flight"] >= 1: + return "unknown_capacity_trial_in_flight" + return None + + +def _profile_snapshot( + profile: dict[str, Any], reference: dict[str, str], binding: dict[str, Any] | None, +) -> dict[str, Any]: + frozen = json.loads(canonical_json(profile)) + return { + "reference": dict(reference), + "binding": dict(binding) if binding is not None else None, + "profile_id": profile["id"], + "profile_sha256": hashlib.sha256(canonical_json(profile).encode()).hexdigest(), + "profile": frozen, + } + + +def _policy_candidates( + role: str, + policy: dict[str, Any], + profiles: dict[str, dict[str, Any]], + bindings: dict[str, dict[str, Any]], + minimum_quality: str, + capacity: dict[str, dict[str, Any]], + implementer: dict[str, Any] | None, +) -> tuple[list[dict[str, Any]], list[dict[str, Any]], bool]: + references = list(policy["roles"].get(role, [])) + if role == "reviewer" and implementer is not None and policy.get( + "prefer_different_harness_for_review", False, + ): + references.sort( + key=lambda reference: ( + (_resolve_reference(reference, profiles, bindings)[0] or {}).get("harness") + == implementer["harness"] + ) + ) + eligible, excluded, has_static_candidate = [], [], False + seen = set() + for reference in references: + profile, binding, resolution_error = _resolve_reference( + reference, profiles, bindings, + ) + reason = resolution_error + if profile is not None and reason is None: + reason = _static_reason( + profile, + role, + minimum_quality, + policy["account_pools"].get(profile["account_pool_id"]), + implementer, + policy["require_different_model_for_review"], + ) + if reason is None: + has_static_candidate = True + reason = _capacity_reason(profile, capacity) + profile_id = profile["id"] if profile is not None else None + if reason is None and profile_id in seen: + reason = "duplicate_effective_profile" + if reason is not None: + excluded.append({ + "reference": dict(reference), + "profile_id": profile_id, + "reason": reason, + }) + continue + seen.add(profile_id) + eligible.append(_profile_snapshot(profile, reference, binding)) + return eligible, excluded, has_static_candidate + + +def resolve_routing( + task: dict[str, Any], + profile_registry: dict[str, Any], + policy: dict[str, Any], + *, + availability: dict[str, Any] | None = None, + profiles_sha256: str | None = None, + policy_sha256: str | None = None, +) -> dict[str, Any]: + """Resolve and freeze every model role used by a fixed workflow.""" + validate_task(task) + validate_profile_registry(profile_registry) + validate_policy(policy) + profiles = {profile["id"]: profile for profile in profile_registry["profiles"]} + bindings = profile_registry["bindings"] + capacity = _availability_snapshot(policy, availability) + minimum_quality = policy["task_classes"].get(task["task_class"]) + if minimum_quality is None: + raise PolicyDenied(f"policy does not authorize task class: {task['task_class']}") + + roles = list(WORKFLOW_ROLES[task["workflow"]]) + if task["lead"]["mode"] == "headless": + roles.append("lead") + overrides = task["routing"].get("overrides", {}) + unsupported_overrides = set(overrides) - set(roles) + if unsupported_overrides: + raise ContractError( + f"routing overrides are not roles in {task['workflow']}: " + f"{sorted(unsupported_overrides)}" + ) + missing_roles = [role for role in roles if not policy["roles"].get(role)] + if missing_roles: + raise PolicyDenied(f"policy has no candidates for required roles: {missing_roles}") + + selected_roles: dict[str, Any] = {} + implementer = None + max_fallbacks = task["budget"]["max_fallbacks_per_step"] + for role in roles: + eligible, excluded, has_static_candidate = _policy_candidates( + role, + policy, + profiles, + bindings, + minimum_quality, + capacity, + implementer, + ) + override = overrides.get(role) + selected = None + source = "automatic" + fallback_mode = "policy" + if override is not None: + source = "override" + fallback_mode = override.get("fallback", "none") + pinned = profiles.get(override["profile_id"]) + if pinned is None: + raise ProfileUnsupported( + f"pinned {role} profile does not exist: {override['profile_id']}" + ) + static_reason = _static_reason( + pinned, + role, + minimum_quality, + policy["account_pools"].get(pinned["account_pool_id"]), + implementer, + policy["require_different_model_for_review"], + ) + if static_reason is not None: + raise ProfileUnsupported( + f"pinned {role} profile is unsupported: {static_reason}" + ) + capacity_reason = _capacity_reason(pinned, capacity) + if capacity_reason is None: + selected = _profile_snapshot( + pinned, {"kind": "profile", "id": pinned["id"]}, None, + ) + elif fallback_mode == "none": + raise CapabilityUnavailable( + f"pinned {role} profile is unavailable: {capacity_reason}" + ) + else: + excluded.insert(0, { + "reference": {"kind": "profile", "id": pinned["id"]}, + "profile_id": pinned["id"], + "reason": capacity_reason, + }) + if selected is None: + if not eligible: + if has_static_candidate: + raise CapabilityUnavailable( + f"no currently available profile for role: {role}" + ) + raise PolicyDenied(f"no policy-eligible profile for role: {role}") + selected = eligible.pop(0) + fallbacks = [] + if fallback_mode == "policy": + fallbacks = [ + candidate for candidate in eligible + if candidate["profile_id"] != selected["profile_id"] + ][:max_fallbacks] + selected_roles[role] = { + "source": source, + "fallback_mode": fallback_mode, + "selected": selected, + "fallbacks": fallbacks, + "excluded": excluded, + } + if role == "implementer": + implementer = selected["profile"] + + return { + "schema_version": 1, + "policy": { + "id": policy["id"], + "version": policy["version"], + "sha256": policy_sha256 or hashlib.sha256(canonical_json(policy).encode()).hexdigest(), + }, + "profile_registry": { + "schema_version": profile_registry["schema_version"], + "sha256": profiles_sha256 + or hashlib.sha256(canonical_json(profile_registry).encode()).hexdigest(), + }, + "task_class": task["task_class"], + "workflow": task["workflow"], + "capacity": capacity, + "roles": selected_roles, + } + + +def load_routing( + task: dict[str, Any], + profiles_payload: bytes | str, + policy_payload: bytes | str, + *, + availability: dict[str, Any] | None = None, +) -> dict[str, Any]: + """Strictly decode profile/policy files and freeze hashes with selection.""" + profile_registry, profiles_sha256 = _strict_json(profiles_payload, "profiles file") + policy, policy_sha256 = _strict_json(policy_payload, "policy file") + return resolve_routing( + task, + profile_registry, + policy, + availability=availability, + profiles_sha256=profiles_sha256, + policy_sha256=policy_sha256, + ) diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py index 5879b8c..663af77 100644 --- a/plugin/core/src/devsquad/validation.py +++ b/plugin/core/src/devsquad/validation.py @@ -96,6 +96,35 @@ def validate_profile(value: dict[str, Any]) -> None: if not isinstance(value["quality_status"], str) or value["quality_status"] not in {"unvalidated", "trial", "proven", "suspended"}: raise ContractError("invalid quality status") +def validate_profile_registry(value: dict[str, Any]) -> None: + fields = {"schema_version", "profiles", "bindings"} + _exact(value, fields, fields, "profile registry") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("invalid profile registry schema version") + profiles = value["profiles"] + if not isinstance(profiles, list) or not profiles: + raise ContractError("profile registry profiles must be a non-empty array") + identifiers = [] + for profile in profiles: + validate_profile(profile) + identifiers.append(profile["id"]) + if len(set(identifiers)) != len(identifiers): + raise ContractError("profile registry profile ids must be unique") + bindings = value["bindings"] + if not isinstance(bindings, dict): + raise ContractError("profile registry bindings must be an object") + for alias, binding in bindings.items(): + if not isinstance(alias, str) or not alias: + raise ContractError("profile binding aliases must be non-empty strings") + _exact(binding, {"profile_id", "version"}, {"profile_id", "version"}, "profile binding") + if not isinstance(binding["profile_id"], str) or not binding["profile_id"]: + raise ContractError("profile binding profile_id must be non-empty") + if type(binding["version"]) is not int or binding["version"] < 1: + raise ContractError("profile binding version must be positive") + if binding["profile_id"] not in identifiers: + raise ContractError(f"profile binding target does not exist: {binding['profile_id']}") + + def validate_policy(value: dict[str, Any]) -> None: fields = {"schema_version", "id", "version", "roles", "task_classes", "require_different_model_for_review", "prefer_different_harness_for_review", "account_pools", "experiment_budget"} required = fields - {"prefer_different_harness_for_review"} @@ -116,6 +145,23 @@ def validate_policy(value: dict[str, Any]) -> None: raise ContractError("task_classes must map names to quality status") if not all(isinstance(k, str) and k and isinstance(v, dict) for k, v in value["account_pools"].items()): raise ContractError("account_pools must map names to objects") + for pool in value["account_pools"].values(): + _exact( + pool, + {"allowed_billing_modes", "max_concurrency", "unknown_capacity_policy"}, + {"allowed_billing_modes", "max_concurrency"}, + "account pool policy", + ) + modes = pool["allowed_billing_modes"] + if (not isinstance(modes, list) or not modes + or len(set(modes)) != len(modes) + or not all(isinstance(mode, str) and mode in {"subscription", "paid_api"} + for mode in modes)): + raise ContractError("account pool billing modes are invalid") + if type(pool["max_concurrency"]) is not int or pool["max_concurrency"] < 1: + raise ContractError("account pool max_concurrency must be positive") + if pool.get("unknown_capacity_policy", "allow_bounded") not in {"allow_bounded", "block"}: + raise ContractError("account pool unknown_capacity_policy is invalid") if not all(isinstance(k, str) and k and type(v) is int and v >= 0 for k, v in value["experiment_budget"].items()): raise ContractError("experiment_budget must contain non-negative integers") diff --git a/test/core/test_router.py b/test/core/test_router.py new file mode 100644 index 0000000..2674553 --- /dev/null +++ b/test/core/test_router.py @@ -0,0 +1,360 @@ +import copy +import hashlib +import json +from pathlib import Path +import sys +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ( + CapabilityUnavailable, + ContractError, + PolicyDenied, + ProfileUnsupported, +) +from devsquad.router import load_routing, resolve_routing +from devsquad.validation import validate_policy, validate_profile_registry + + +def profile( + profile_id, + *, + harness="codex", + family="family-a", + model=None, + permission="read_only", + pool="pool-a", + billing="subscription", + quality="proven", +): + return { + "id": profile_id, + "harness": harness, + "model_family": family, + "model_id": model or f"model-{profile_id}", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": permission, + "account_pool_id": pool, + "billing_mode": billing, + "quality_status": quality, + "evidence_refs": [f"evidence-{profile_id}"], + } + + +def pool(*, modes=None, concurrency=2, unknown="allow_bounded"): + return { + "allowed_billing_modes": modes or ["subscription"], + "max_concurrency": concurrency, + "unknown_capacity_policy": unknown, + } + + +class RouterTest(unittest.TestCase): + def setUp(self): + self.task = { + "schema_version": 1, + "project": { + "repo_path": "/tmp/router-fixture", + "base_ref": "base", + "target_ref": "target", + }, + "workflow": "branch-review", + "goal": "Review the frozen change.", + "task_class": "fixture-review", + "acceptance": [{ + "id": "review", + "description": "Return a bound review.", + "evidence_kind": "review", + }], + "checks": [], + "scope": {"read_paths": ["src"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": 300, + "max_worker_invocations": 3, + "max_revisions": 0, + "max_fallbacks_per_step": 1, + }, + "origin": {"surface": "test"}, + } + self.registry = { + "schema_version": 1, + "profiles": [ + profile("review-a"), + profile("review-b", harness="claude", family="family-b"), + ], + "bindings": { + "review.deep": {"profile_id": "review-b", "version": 7}, + }, + } + self.policy = { + "schema_version": 1, + "id": "fixture-policy", + "version": 3, + "roles": { + "reviewer": [ + {"kind": "alias", "id": "review.deep"}, + {"kind": "profile", "id": "review-a"}, + ], + }, + "task_classes": {"fixture-review": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": {"pool-a": pool()}, + "experiment_budget": {}, + } + + def test_alias_selection_is_deterministic_frozen_and_explained(self): + first = resolve_routing(self.task, self.registry, self.policy) + second = resolve_routing(self.task, self.registry, self.policy) + self.assertEqual(first, second) + reviewer = first["roles"]["reviewer"] + self.assertEqual(reviewer["source"], "automatic") + self.assertEqual(reviewer["selected"]["profile_id"], "review-b") + self.assertEqual( + reviewer["selected"]["binding"], + {"alias": "review.deep", "version": 7, "profile_id": "review-b"}, + ) + self.assertEqual( + [candidate["profile_id"] for candidate in reviewer["fallbacks"]], + ["review-a"], + ) + self.assertEqual(first["capacity"]["pool-a"]["status"], "unknown") + self.assertEqual( + first["capacity"]["pool-a"]["unknown_capacity_policy"], + "allow_bounded", + ) + + self.registry["profiles"][1]["model_id"] = "mutated-after-selection" + self.registry["bindings"]["review.deep"]["version"] = 8 + self.assertNotEqual( + reviewer["selected"]["profile"]["model_id"], + "mutated-after-selection", + ) + self.assertEqual(reviewer["selected"]["binding"]["version"], 7) + + def test_one_role_pin_leaves_headless_lead_automatic(self): + pinned = profile("review-pin", harness="grok", family="family-c") + lead = profile("lead-a", harness="claude", family="family-b") + self.registry["profiles"].extend([pinned, lead]) + self.policy["roles"]["lead"] = [{"kind": "profile", "id": "lead-a"}] + self.task["lead"] = {"mode": "headless"} + self.task["routing"]["overrides"] = { + "reviewer": {"profile_id": "review-pin", "fallback": "none"}, + } + routed = resolve_routing(self.task, self.registry, self.policy) + self.assertEqual(routed["roles"]["reviewer"]["source"], "override") + self.assertEqual( + routed["roles"]["reviewer"]["selected"]["profile_id"], "review-pin", + ) + self.assertEqual(routed["roles"]["reviewer"]["fallbacks"], []) + self.assertEqual(routed["roles"]["lead"]["source"], "automatic") + self.assertEqual(routed["roles"]["lead"]["selected"]["profile_id"], "lead-a") + + self.task["routing"]["overrides"]["implementer"] = { + "profile_id": "review-pin", + } + with self.assertRaisesRegex(ContractError, "not roles in branch-review"): + resolve_routing(self.task, self.registry, self.policy) + + def test_missing_or_statically_invalid_pin_never_falls_back(self): + self.task["routing"]["overrides"] = { + "reviewer": {"profile_id": "missing", "fallback": "policy"}, + } + with self.assertRaisesRegex(ProfileUnsupported, "does not exist"): + resolve_routing(self.task, self.registry, self.policy) + + writer = profile("writer", permission="workspace_write") + self.registry["profiles"].append(writer) + self.task["routing"]["overrides"]["reviewer"]["profile_id"] = "writer" + with self.assertRaisesRegex(ProfileUnsupported, "permission_mismatch"): + resolve_routing(self.task, self.registry, self.policy) + + def test_unavailable_pin_obeys_explicit_fallback_mode(self): + self.registry["profiles"][1]["account_pool_id"] = "pool-b" + self.policy["account_pools"]["pool-b"] = pool() + availability = { + "pool-a": {"status": "available", "in_flight": 0}, + "pool-b": {"status": "exhausted", "in_flight": 0}, + } + self.task["routing"]["overrides"] = { + "reviewer": {"profile_id": "review-b", "fallback": "none"}, + } + with self.assertRaisesRegex(CapabilityUnavailable, "account_pool_exhausted"): + resolve_routing( + self.task, self.registry, self.policy, availability=availability, + ) + + self.task["routing"]["overrides"]["reviewer"]["fallback"] = "policy" + routed = resolve_routing( + self.task, self.registry, self.policy, availability=availability, + ) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "review-a") + self.assertEqual(reviewer["fallback_mode"], "policy") + self.assertEqual(reviewer["excluded"][0]["profile_id"], "review-b") + self.assertEqual(reviewer["excluded"][0]["reason"], "account_pool_exhausted") + + def test_static_policy_filters_are_recorded_without_relaxation(self): + candidates = [ + profile("suspended", quality="suspended"), + profile("trial", quality="trial"), + profile("writer", permission="workspace_write"), + profile("paid", billing="paid_api"), + profile("eligible"), + ] + self.registry = {"schema_version": 1, "profiles": candidates, "bindings": {}} + self.policy["roles"]["reviewer"] = [ + {"kind": "profile", "id": candidate["id"]} for candidate in candidates + ] + routed = resolve_routing(self.task, self.registry, self.policy) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "eligible") + self.assertEqual( + [item["reason"] for item in reviewer["excluded"]], + [ + "profile_suspended", + "quality_below_task_minimum", + "permission_mismatch", + "billing_mode_not_allowed", + ], + ) + + def test_delivery_review_is_different_model_and_prefers_different_harness(self): + implementer = profile( + "implementer", + model="shared-model", + permission="workspace_write", + ) + same_model = profile( + "same-model", + harness="claude", + model="shared-model", + ) + same_harness = profile("same-harness", model="other-codex-model") + other_harness = profile( + "other-harness", harness="claude", family="family-b", model="other-model", + ) + self.registry = { + "schema_version": 1, + "profiles": [implementer, same_model, same_harness, other_harness], + "bindings": {}, + } + self.policy["roles"] = { + "implementer": [{"kind": "profile", "id": "implementer"}], + "reviewer": [ + {"kind": "profile", "id": "same-model"}, + {"kind": "profile", "id": "same-harness"}, + {"kind": "profile", "id": "other-harness"}, + ], + } + self.task["workflow"] = "issue-delivery" + self.task["scope"]["write_paths"] = ["src"] + routed = resolve_routing(self.task, self.registry, self.policy) + self.assertEqual( + routed["roles"]["implementer"]["selected"]["profile_id"], + "implementer", + ) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "other-harness") + reasons = {item["profile_id"]: item["reason"] for item in reviewer["excluded"]} + self.assertEqual(reasons["same-model"], "review_model_not_independent") + self.assertEqual( + [candidate["profile_id"] for candidate in reviewer["fallbacks"]], + ["same-harness"], + ) + + def test_typed_unknown_and_concurrency_capacity_are_fail_closed(self): + self.policy["account_pools"]["pool-a"] = pool(unknown="block") + with self.assertRaisesRegex(CapabilityUnavailable, "currently available"): + resolve_routing(self.task, self.registry, self.policy) + + self.policy["account_pools"]["pool-a"] = pool( + concurrency=3, unknown="allow_bounded", + ) + with self.assertRaisesRegex(CapabilityUnavailable, "currently available"): + resolve_routing( + self.task, + self.registry, + self.policy, + availability={"pool-a": {"status": "unknown", "in_flight": 1}}, + ) + routed = resolve_routing( + self.task, + self.registry, + self.policy, + availability={"pool-a": {"status": "unknown", "in_flight": 0}}, + ) + self.assertEqual(routed["roles"]["reviewer"]["selected"]["profile_id"], "review-b") + + with self.assertRaisesRegex(CapabilityUnavailable, "currently available"): + resolve_routing( + self.task, + self.registry, + self.policy, + availability={"pool-a": {"status": "available", "in_flight": 3}}, + ) + + def test_policy_missing_task_class_or_required_role_is_denied(self): + del self.policy["task_classes"]["fixture-review"] + with self.assertRaisesRegex(PolicyDenied, "does not authorize task class"): + resolve_routing(self.task, self.registry, self.policy) + self.policy["task_classes"]["fixture-review"] = "proven" + del self.policy["roles"]["reviewer"] + with self.assertRaisesRegex(PolicyDenied, "required roles"): + resolve_routing(self.task, self.registry, self.policy) + + def test_loader_rejects_duplicate_keys_nan_and_hashes_exact_bytes(self): + profiles_bytes = (json.dumps(self.registry, indent=2) + "\n").encode() + policy_bytes = (json.dumps(self.policy, separators=(",", ":")) + "\n").encode() + routed = load_routing(self.task, profiles_bytes, policy_bytes) + self.assertEqual( + routed["profile_registry"]["sha256"], hashlib.sha256(profiles_bytes).hexdigest(), + ) + self.assertEqual( + routed["policy"]["sha256"], hashlib.sha256(policy_bytes).hexdigest(), + ) + with self.assertRaisesRegex(ContractError, "duplicate key"): + load_routing( + self.task, + b'{"schema_version":1,"schema_version":1,"profiles":[],"bindings":{}}', + policy_bytes, + ) + with self.assertRaisesRegex(ContractError, "non-finite"): + load_routing( + self.task, + b'{"schema_version":1,"profiles":[],"bindings":{},"bad":NaN}', + policy_bytes, + ) + with self.assertRaisesRegex(ContractError, "UTF-8 JSON"): + load_routing(self.task, b"\xff", policy_bytes) + + def test_registry_and_account_pool_shapes_are_strict(self): + duplicate = copy.deepcopy(self.registry) + duplicate["profiles"].append(copy.deepcopy(duplicate["profiles"][0])) + with self.assertRaisesRegex(ContractError, "ids must be unique"): + validate_profile_registry(duplicate) + missing_target = copy.deepcopy(self.registry) + missing_target["bindings"]["review.deep"]["profile_id"] = "missing" + with self.assertRaisesRegex(ContractError, "target does not exist"): + validate_profile_registry(missing_target) + + invalid_policy = copy.deepcopy(self.policy) + invalid_policy["account_pools"]["pool-a"]["extra"] = True + with self.assertRaisesRegex(ContractError, "account pool policy fields invalid"): + validate_policy(invalid_policy) + invalid_policy = copy.deepcopy(self.policy) + invalid_policy["account_pools"]["pool-a"]["max_concurrency"] = True + with self.assertRaisesRegex(ContractError, "max_concurrency"): + validate_policy(invalid_policy) + + +if __name__ == "__main__": + unittest.main() From 1c7b61481475bf62b3e7158e98ab73e7632b1719 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 16 Sep 2026 00:57:38 +0530 Subject: [PATCH 048/197] feat: freeze M3 routing during preflight --- plugin/core/src/devsquad/service.py | 14 +++++++- test/core/devsquad_test_fixtures.py | 44 +++++++++++++++++++++++ test/core/test_m2_cross_process.py | 6 ++-- test/core/test_service.py | 55 +++++++++++++++++++++++++++-- 4 files changed, 113 insertions(+), 6 deletions(-) create mode 100644 test/core/devsquad_test_fixtures.py diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index fe1a35e..7f49d05 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -14,6 +14,7 @@ from typing import Any from .contracts import ContractError +from .router import load_routing from .store import ( ConflictError, HandoffClaim, @@ -158,15 +159,26 @@ def oid(ref: str) -> str: if result.returncode != 0: raise ContractError(f"Git ref does not resolve to a commit: {ref}") return result.stdout.strip() configs = {} + config_payloads = {} for label in ("profiles_file", "policy_file"): candidate = Path(task["routing"][label]) path = (repo / candidate).resolve() if not candidate.is_absolute() else candidate.resolve() if path != repo and repo not in path.parents: raise ContractError(f"{label} escapes project") - data = path.read_bytes(); configs[label] = {"path": str(path), "sha256": hashlib.sha256(data).hexdigest()} + data = path.read_bytes() + config_payloads[label] = data + configs[label] = { + "path": str(path), "sha256": hashlib.sha256(data).hexdigest(), + } snapshot = {"task": task, "base_oid": oid(task["project"]["base_ref"]), "target_oid": oid(task["project"]["target_ref"]), "configs": configs} if internal_delay is not None: if internal_delay < 0 or internal_delay > 60: raise ContractError("internal fake delay is invalid") snapshot["internal_fake_delay"] = internal_delay + else: + snapshot["routing"] = load_routing( + task, + config_payloads["profiles_file"], + config_payloads["policy_file"], + ) return snapshot def _continue_preparation( diff --git a/test/core/devsquad_test_fixtures.py b/test/core/devsquad_test_fixtures.py new file mode 100644 index 0000000..54c16ee --- /dev/null +++ b/test/core/devsquad_test_fixtures.py @@ -0,0 +1,44 @@ +import json + + +def branch_review_routing_documents(): + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "fixture-reviewer", + "harness": "fixture", + "model_family": "fixture-family-a", + "model_id": "fixture-review-model", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "fixture-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["tracked-fixture"], + }], + "bindings": { + "review.deep": {"profile_id": "fixture-reviewer", "version": 1}, + }, + } + policy = { + "schema_version": 1, + "id": "fixture-policy", + "version": 1, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"fixture-review-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "fixture-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + return ( + json.dumps(profiles, sort_keys=True) + "\n", + json.dumps(policy, sort_keys=True) + "\n", + ) diff --git a/test/core/test_m2_cross_process.py b/test/core/test_m2_cross_process.py index 5f5dde5..9c3c8ea 100644 --- a/test/core/test_m2_cross_process.py +++ b/test/core/test_m2_cross_process.py @@ -18,6 +18,7 @@ from devsquad.service import Service from devsquad.store import Store, request_hash from devsquad.supervisor import inspect_process +from devsquad_test_fixtures import branch_review_routing_documents def service_start(runtime, task, key, barrier, results): @@ -84,8 +85,9 @@ def setUp(self): subprocess.run(["git", "init", "-q", str(self.repo)], check=True) subprocess.run(["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], check=True) subprocess.run(["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True) - (self.repo / "profiles.json").write_text("{}\n") - (self.repo / "policy.json").write_text("{}\n") + profiles_json, policy_json = branch_review_routing_documents() + (self.repo / "profiles.json").write_text(profiles_json) + (self.repo / "policy.json").write_text(policy_json) subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) self.task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) diff --git a/test/core/test_service.py b/test/core/test_service.py index 2238c4d..2013b73 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -1,3 +1,4 @@ +import hashlib import json import multiprocessing import os @@ -17,6 +18,7 @@ from devsquad.service import Service from devsquad.store import ConflictError, Store from devsquad.supervisor import Supervisor, inspect_process +from devsquad_test_fixtures import branch_review_routing_documents def concurrent_receipt_import(database, artifacts, run_id, barrier, results): @@ -88,7 +90,9 @@ def setUp(self): subprocess.run(["git", "init", "-q", str(self.repo)], check=True) subprocess.run(["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], check=True) subprocess.run(["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True) - (self.repo / "profiles.json").write_text("{}\n"); (self.repo / "policy.json").write_text("{}\n") + self.profiles_json, self.policy_json = branch_review_routing_documents() + (self.repo / "profiles.json").write_text(self.profiles_json) + (self.repo / "policy.json").write_text(self.policy_json) subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) self.task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) @@ -485,6 +489,51 @@ def test_pre_attempt_failure_and_cancellation_have_durable_receipts(self): self.assertTrue(cancellation_receipt["cancelled"]) self.assertEqual(cancellation_receipt["phase"],"preparing") + def test_public_preflight_freezes_profile_policy_and_alias_selection(self): + started=self.service.start(self.task,"frozen-routing") + self.assertEqual(started["error"]["error"],"CAPABILITY_UNAVAILABLE") + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + snapshot=json.loads(store.run(started["run_id"])["mutable_snapshot"]) + finally: + store.close() + routed=snapshot["routing"] + reviewer=routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"],"fixture-reviewer") + self.assertEqual( + reviewer["selected"]["binding"], + {"alias":"review.deep","profile_id":"fixture-reviewer","version":1}, + ) + self.assertEqual( + routed["profile_registry"]["sha256"], + hashlib.sha256(self.profiles_json.encode()).hexdigest(), + ) + self.assertEqual( + routed["policy"]["sha256"], + hashlib.sha256(self.policy_json.encode()).hexdigest(), + ) + self.assertEqual( + snapshot["configs"]["profiles_file"]["sha256"], + routed["profile_registry"]["sha256"], + ) + self.assertEqual( + snapshot["configs"]["policy_file"]["sha256"], + routed["policy"]["sha256"], + ) + + (self.repo/"profiles.json").write_text("not valid after snapshot\n") + (self.repo/"policy.json").write_text("also changed\n") + replayed=self.service.start(self.task,"frozen-routing") + self.assertEqual(replayed["run_id"],started["run_id"]) + self.assertFalse(replayed["created"]) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + self.assertEqual( + json.loads(store.run(started["run_id"])["mutable_snapshot"]),snapshot, + ) + finally: + store.close() + def test_invalid_predecessor_fails_with_run_context_and_receipt(self): started=self.service.start( self.task,"invalid-predecessor","does-not-exist",_internal_fake_delay=.01, @@ -515,8 +564,8 @@ def test_public_active_and_foreign_predecessor_matrix(self): subprocess.run(["git","init","-q",str(other)],check=True) subprocess.run(["git","-C",str(other),"config","user.email","test@example.invalid"],check=True) subprocess.run(["git","-C",str(other),"config","user.name","Test"],check=True) - (other/"profiles.json").write_text("{}\n") - (other/"policy.json").write_text("{}\n") + (other/"profiles.json").write_text(self.profiles_json) + (other/"policy.json").write_text(self.policy_json) subprocess.run(["git","-C",str(other),"add","."],check=True) subprocess.run(["git","-C",str(other),"commit","-qm","foreign"],check=True) foreign_task=json.loads(json.dumps(self.task)) From cd9a881c9cf7b3128b0ffa8d19a60e32f6029616 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 16 Sep 2026 01:10:37 +0530 Subject: [PATCH 049/197] feat: freeze M3 review workspaces --- plugin/core/src/devsquad/service.py | 66 +++++-- plugin/core/src/devsquad/store.py | 33 +++- plugin/core/src/devsquad/workspaces.py | 248 +++++++++++++++++++++++++ test/core/test_service.py | 128 ++++++++++++- test/core/test_store.py | 57 ++++++ test/core/test_workspaces.py | 190 +++++++++++++++++++ 6 files changed, 706 insertions(+), 16 deletions(-) create mode 100644 plugin/core/src/devsquad/workspaces.py create mode 100644 test/core/test_workspaces.py diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 7f49d05..54cf7ea 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -24,6 +24,13 @@ canonical_json, ) from .validation import validate_task +from .workspaces import ( + assert_clean_inputs, + committed_regular_file, + prepare_review_workspace, + repo_relative_config, + resolve_commit, +) class Service: @@ -147,29 +154,45 @@ def _handoff_payload(snapshot: HandoffSnapshot, *, include_packet: bool) -> dict payload["claim_expires_at"] = None return payload - @staticmethod def _resolve_snapshot( + self, task: dict[str, Any], internal_delay: float | None, resolved_repo: Path | None = None, + *, + project_id: str | None = None, + run_id: str | None = None, ) -> dict[str, Any]: repo = resolved_repo or Path(task["project"]["repo_path"]).resolve(strict=True) - def oid(ref: str) -> str: - result = subprocess.run(["git", "-C", str(repo), "rev-parse", "--verify", f"{ref}^{{commit}}"], text=True, capture_output=True, check=False) - if result.returncode != 0: raise ContractError(f"Git ref does not resolve to a commit: {ref}") - return result.stdout.strip() + base_oid = resolve_commit(repo, task["project"]["base_ref"]) + target_oid = resolve_commit(repo, task["project"]["target_ref"]) + scope_paths = tuple(dict.fromkeys( + task["scope"]["read_paths"] + task["scope"]["write_paths"] + )) + config_paths = { + label: repo_relative_config(repo, task["routing"][label], label) + for label in ("profiles_file", "policy_file") + } + if internal_delay is None: + assert_clean_inputs(repo, scope_paths, config_paths.values()) configs = {} config_payloads = {} - for label in ("profiles_file", "policy_file"): - candidate = Path(task["routing"][label]) - path = (repo / candidate).resolve() if not candidate.is_absolute() else candidate.resolve() - if path != repo and repo not in path.parents: raise ContractError(f"{label} escapes project") - data = path.read_bytes() + for label, relative_path in config_paths.items(): + path = repo / relative_path + data = ( + committed_regular_file(repo, target_oid, relative_path) + if internal_delay is None else path.read_bytes() + ) config_payloads[label] = data configs[label] = { "path": str(path), "sha256": hashlib.sha256(data).hexdigest(), } - snapshot = {"task": task, "base_oid": oid(task["project"]["base_ref"]), "target_oid": oid(task["project"]["target_ref"]), "configs": configs} + snapshot = { + "task": task, + "base_oid": base_oid, + "target_oid": target_oid, + "configs": configs, + } if internal_delay is not None: if internal_delay < 0 or internal_delay > 60: raise ContractError("internal fake delay is invalid") snapshot["internal_fake_delay"] = internal_delay @@ -179,6 +202,18 @@ def oid(ref: str) -> str: config_payloads["profiles_file"], config_payloads["policy_file"], ) + if project_id is None or run_id is None: + raise ContractError("public preflight requires run-owned workspace identity") + snapshot["workspace"] = prepare_review_workspace( + repo, + self.runtime, + project_id, + run_id, + base_oid, + target_oid, + scope_paths, + required_clean_paths=config_paths.values(), + ) return snapshot def _continue_preparation( @@ -200,7 +235,14 @@ def _continue_preparation( worktree = store.preparation_worktree( run_id, fencing_token, Path(task["project"]["repo_path"]), ) - snapshot = self._resolve_snapshot(task, internal_delay, worktree) + project_id = store.run(run_id)["project_id"] + snapshot = self._resolve_snapshot( + task, + internal_delay, + worktree, + project_id=project_id, + run_id=run_id, + ) if internal_delay is None: error = { "error": "CAPABILITY_UNAVAILABLE", diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 6529e3a..b09f092 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -358,23 +358,50 @@ def _reference_terminal_receipt( ) return version - def complete_preparation(self, run_id: str, fencing_token: int, mutable_snapshot: Any, *, package_path: str | None = None, package_digest: str | None = None, supersedes_run_id: str | None = None) -> int: + def complete_preparation( + self, + run_id: str, + fencing_token: int, + mutable_snapshot: Any, + *, + package_path: str | None = None, + package_digest: str | None = None, + supersedes_run_id: str | None = None, + worktree_path: str | None = None, + ) -> int: snapshot = canonical_json(mutable_snapshot) + resolved_worktree = None + worktree_common = None + if worktree_path is not None: + if not isinstance(worktree_path, str) or not worktree_path or not Path(worktree_path).is_absolute(): + raise ContractError("prepared worktree path must be absolute") + resolved_worktree = str(Path(worktree_path).resolve(strict=True)) + worktree_common = str(git_common_dir(Path(resolved_worktree))) self.connection.execute("BEGIN IMMEDIATE") try: row = self.connection.execute( - "SELECT r.state,r.phase,r.version,c.fencing_token,c.active FROM runs r JOIN claims c ON c.run_id=r.id WHERE r.id=?", + "SELECT r.state,r.phase,r.version,c.fencing_token,c.active," + "p.git_common_dir FROM runs r JOIN claims c ON c.run_id=r.id " + "JOIN projects p ON p.id=r.project_id WHERE r.id=?", (run_id,), ).fetchone() if not row or row["state"] != "queued" or row["phase"] != "preparing" or not row["active"] or row["fencing_token"] != fencing_token: raise ConflictError("preparation claim is stale or cancelled") + if worktree_common is not None and worktree_common != row["git_common_dir"]: + raise ContractError("prepared worktree belongs to a different project") if supersedes_run_id is not None: predecessor = self.connection.execute("SELECT project_id,state FROM runs WHERE id=?", (supersedes_run_id,)).fetchone() project = self.connection.execute("SELECT project_id FROM runs WHERE id=?", (run_id,)).fetchone() if not predecessor or predecessor["project_id"] != project["project_id"] or predecessor["state"] not in TERMINAL_STATES: raise ConflictError("superseded run must be terminal and belong to the same project") version, now = row["version"] + 1, _utc_now() - self.connection.execute("UPDATE runs SET mutable_snapshot=?,package_path=?,package_digest=?,supersedes_run_id=?,phase=NULL,version=?,updated_at=? WHERE id=?", (snapshot, package_path, package_digest, supersedes_run_id, version, now, run_id)) + self.connection.execute( + "UPDATE runs SET mutable_snapshot=?,package_path=?,package_digest=?," + "supersedes_run_id=?,worktree_path=COALESCE(?,worktree_path),phase=NULL," + "version=?,updated_at=? WHERE id=?", + (snapshot, package_path, package_digest, supersedes_run_id, + resolved_worktree, version, now, run_id), + ) self.connection.execute("UPDATE claims SET active=0 WHERE run_id=?", (run_id,)) self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.queued','{}',?)", (run_id, version, now)) self.connection.execute("COMMIT") diff --git a/plugin/core/src/devsquad/workspaces.py b/plugin/core/src/devsquad/workspaces.py new file mode 100644 index 0000000..b82be38 --- /dev/null +++ b/plugin/core/src/devsquad/workspaces.py @@ -0,0 +1,248 @@ +"""Frozen Git input and isolated review-workspace preparation for M3.""" + +from __future__ import annotations + +import hashlib +import os +from pathlib import Path, PurePosixPath +import subprocess +from typing import Iterable + +from .contracts import ContractError +from .store import canonical_json, git_common_dir + + +def _git(repo: Path, *args: str) -> bytes: + try: + result = subprocess.run( + ["git", "-C", str(repo), *args], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + check=False, + ) + except OSError as exc: + raise ContractError("Git is unavailable while preparing the review workspace") from exc + if result.returncode != 0: + detail = result.stderr.decode("utf-8", "replace").strip() + raise ContractError(f"Git workspace operation failed: {detail or args[0]}") + return result.stdout + + +def resolve_commit(repo: Path, ref: str) -> str: + if not isinstance(ref, str) or not ref: + raise ContractError("Git ref must be a non-empty string") + try: + result = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "--verify", f"{ref}^{{commit}}"], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + check=False, + ) + except OSError as exc: + raise ContractError("Git is unavailable while resolving a commit") from exc + if result.returncode != 0: + raise ContractError(f"Git ref does not resolve to a commit: {ref}") + oid = result.stdout.decode().strip() + if len(oid) != 40 or any(character not in "0123456789abcdef" for character in oid): + raise ContractError(f"Git ref did not resolve to a full commit OID: {ref}") + return oid + + +def _decode_paths(payload: bytes, label: str) -> list[str]: + values = payload.split(b"\0") + if values and values[-1] == b"": + values.pop() + result = [] + for raw in values: + try: + value = raw.decode("utf-8") + except UnicodeDecodeError as exc: + raise ContractError(f"{label} contains a non-UTF-8 Git path") from exc + if not value: + raise ContractError(f"{label} contains an empty Git path") + result.append(value) + return result + + +def dirty_paths(repo: Path) -> tuple[str, ...]: + paths = set() + for args in ( + ("diff", "--no-renames", "--name-only", "-z", "--"), + ("diff", "--cached", "--no-renames", "--name-only", "-z", "--"), + ("ls-files", "--others", "--exclude-standard", "-z", "--"), + ): + paths.update(_decode_paths(_git(repo, *args), "dirty inventory")) + return tuple(sorted(paths)) + + +def _normalized_relative(value: str, label: str) -> str: + if not isinstance(value, str) or not value: + raise ContractError(f"{label} must be a non-empty repository-relative path") + candidate = PurePosixPath(value) + if candidate.is_absolute() or ".." in candidate.parts: + raise ContractError(f"{label} must be repository-relative without traversal") + normalized = candidate.as_posix() + return "." if normalized in {"", "."} else normalized.rstrip("/") + + +def repo_relative_config(repo: Path, configured: str, label: str) -> str: + if not isinstance(configured, str) or not configured: + raise ContractError(f"{label} must be a non-empty path") + source = Path(configured) + if source.is_absolute(): + try: + return source.resolve().relative_to(repo.resolve()).as_posix() + except ValueError as exc: + raise ContractError(f"{label} escapes project") from exc + return _normalized_relative(configured, label) + + +def _intersects(path: str, scope: str) -> bool: + return scope == "." or path == scope or path.startswith(f"{scope}/") + + +def assert_clean_inputs( + repo: Path, + scope_paths: Iterable[str], + required_paths: Iterable[str] = (), +) -> None: + scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) + required = tuple( + _normalized_relative(path, "required committed path") for path in required_paths + ) + intersections = [ + path for path in dirty_paths(repo) + if path in required or any(_intersects(path, scope) for scope in scopes) + ] + if intersections: + raise ContractError( + "committed-input mode rejects dirty scoped/config paths: " + + ", ".join(intersections) + ) + + +def committed_regular_file(repo: Path, commit_oid: str, relative_path: str) -> bytes: + relative = _normalized_relative(relative_path, "committed file path") + listing = _git(repo, "ls-tree", "-z", commit_oid, "--", relative) + records = [record for record in listing.split(b"\0") if record] + if len(records) != 1 or b"\t" not in records[0]: + raise ContractError(f"committed config is missing or ambiguous: {relative}") + metadata, raw_name = records[0].split(b"\t", 1) + fields = metadata.split() + if len(fields) != 3 or fields[0] not in {b"100644", b"100755"} or fields[1] != b"blob": + raise ContractError(f"committed config must be a regular file: {relative}") + try: + stored_name = raw_name.decode("utf-8") + except UnicodeDecodeError as exc: + raise ContractError("committed config path is not UTF-8") from exc + if stored_name != relative: + raise ContractError(f"committed config path mismatch: {relative}") + return _git(repo, "show", f"{commit_oid}:{relative}") + + +def _validate_segment(value: str, label: str) -> str: + if (not isinstance(value, str) or not value or value in {".", ".."} + or "/" in value or "\\" in value or "\0" in value or os.sep in value): + raise ContractError(f"{label} is not a safe path segment") + return value + + +def _validate_workspace( + source_repo: Path, workspace: Path, target_oid: str, scope_paths: Iterable[str], +) -> None: + try: + resolved = workspace.resolve(strict=True) + except OSError as exc: + raise ContractError("review workspace does not exist") from exc + top = Path(_git(resolved, "rev-parse", "--show-toplevel").decode().strip()).resolve() + if top != resolved: + raise ContractError("review workspace top level does not match its run-owned path") + if git_common_dir(resolved) != git_common_dir(source_repo): + raise ContractError("review workspace belongs to a different Git project") + if resolve_commit(resolved, "HEAD") != target_oid: + raise ContractError("existing review workspace targets a different commit") + if _git(resolved, "rev-parse", "--abbrev-ref", "HEAD").strip() != b"HEAD": + raise ContractError("review workspace must use detached HEAD") + if dirty_paths(resolved): + raise ContractError("existing review workspace is dirty") + + for scope in scope_paths: + normalized = _normalized_relative(scope, "scope path") + candidate = resolved if normalized == "." else resolved / normalized + try: + destination = candidate.resolve(strict=False) + except OSError as exc: + raise ContractError(f"scope path cannot be resolved: {normalized}") from exc + if destination != resolved and resolved not in destination.parents: + raise ContractError(f"scope path escapes the review workspace: {normalized}") + symlinks = _git(resolved, "ls-tree", "-r", "-z", "HEAD", "--", normalized) + for record in (item for item in symlinks.split(b"\0") if item): + metadata, raw_name = record.split(b"\t", 1) + if metadata.split()[0] != b"120000": + continue + try: + name = raw_name.decode("utf-8") + destination = (resolved / name).resolve(strict=True) + except (UnicodeDecodeError, OSError) as exc: + raise ContractError("scoped symlink is invalid or broken") from exc + if destination != resolved and resolved not in destination.parents: + raise ContractError(f"scoped symlink escapes the review workspace: {name}") + + +def prepare_review_workspace( + source_repo: Path, + runtime: Path, + project_id: str, + run_id: str, + base_oid: str, + target_oid: str, + scope_paths: Iterable[str], + *, + required_clean_paths: Iterable[str] = (), +) -> dict[str, object]: + """Create or validate one detached, run-owned worktree at the target commit.""" + repo = source_repo.resolve(strict=True) + scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) + assert_clean_inputs(repo, scopes, required_clean_paths) + project = _validate_segment(project_id, "project id") + run = _validate_segment(run_id, "run id") + workspace = ( + runtime.resolve() / "projects" / project / "runs" / run / "review-worktree" + ) + workspace.parent.mkdir(parents=True, exist_ok=True) + if not workspace.exists(): + try: + result = subprocess.run( + ["git", "-C", str(repo), "worktree", "add", "--detach", "--quiet", + str(workspace), target_oid], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + check=False, + ) + except OSError as exc: + raise ContractError( + "Git is unavailable while creating the frozen review workspace" + ) from exc + if result.returncode != 0 and not workspace.exists(): + detail = result.stderr.decode("utf-8", "replace").strip() + raise ContractError(f"could not create frozen review workspace: {detail}") + _validate_workspace(repo, workspace, target_oid, scopes) + changed = _decode_paths( + _git( + repo, "diff", "--no-renames", "--name-only", "-z", + base_oid, target_oid, "--", + ), + "candidate diff", + ) + identity = { + "schema_version": 1, + "base_oid": base_oid, + "target_oid": target_oid, + "changed_paths": sorted(changed), + } + return { + **identity, + "path": str(workspace.resolve()), + "candidate_sha256": hashlib.sha256(canonical_json(identity).encode()).hexdigest(), + "scope": list(scopes), + } diff --git a/test/core/test_service.py b/test/core/test_service.py index 2013b73..ec9596e 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -16,7 +16,7 @@ from devsquad.contracts import ExecutionIdentity, LaunchSpec from devsquad.service import Service -from devsquad.store import ConflictError, Store +from devsquad.store import ConflictError, Store, canonical_json from devsquad.supervisor import Supervisor, inspect_process from devsquad_test_fixtures import branch_review_routing_documents @@ -91,6 +91,10 @@ def setUp(self): subprocess.run(["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], check=True) subprocess.run(["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True) self.profiles_json, self.policy_json = branch_review_routing_documents() + (self.repo / "src").mkdir() + (self.repo / "tests").mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# fixture test\n") (self.repo / "profiles.json").write_text(self.profiles_json) (self.repo / "policy.json").write_text(self.policy_json) subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) @@ -490,6 +494,32 @@ def test_pre_attempt_failure_and_cancellation_have_durable_receipts(self): self.assertEqual(cancellation_receipt["phase"],"preparing") def test_public_preflight_freezes_profile_policy_and_alias_selection(self): + base_oid=subprocess.run( + ["git","-C",str(self.repo),"rev-parse","HEAD"], + check=True,text=True,capture_output=True, + ).stdout.strip() + (self.repo/"src/app.py").write_text("VALUE = 'candidate'\n") + subprocess.run(["git","-C",str(self.repo),"add","src/app.py"],check=True) + subprocess.run( + ["git","-C",str(self.repo),"commit","-qm","candidate"],check=True, + ) + target_oid=subprocess.run( + ["git","-C",str(self.repo),"rev-parse","HEAD"], + check=True,text=True,capture_output=True, + ).stdout.strip() + self.task["project"]["base_ref"]=base_oid + self.task["project"]["target_ref"]=target_oid + (self.repo/"notes.txt").write_text("unrelated local work\n") + before_head=subprocess.run( + ["git","-C",str(self.repo),"rev-parse","HEAD"], + check=True,capture_output=True, + ).stdout + before_status=subprocess.run( + ["git","-C",str(self.repo),"status","--porcelain=v1","-z"], + check=True,capture_output=True, + ).stdout + before_index=hashlib.sha256((self.repo/".git/index").read_bytes()).hexdigest() + started=self.service.start(self.task,"frozen-routing") self.assertEqual(started["error"]["error"],"CAPABILITY_UNAVAILABLE") store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") @@ -520,6 +550,43 @@ def test_public_preflight_freezes_profile_policy_and_alias_selection(self): snapshot["configs"]["policy_file"]["sha256"], routed["policy"]["sha256"], ) + workspace=snapshot["workspace"] + self.assertEqual(workspace["base_oid"],base_oid) + self.assertEqual(workspace["target_oid"],target_oid) + self.assertEqual(workspace["changed_paths"],["src/app.py"]) + identity={ + "schema_version":1, + "base_oid":base_oid, + "target_oid":target_oid, + "changed_paths":["src/app.py"], + } + self.assertEqual( + workspace["candidate_sha256"], + hashlib.sha256(canonical_json(identity).encode()).hexdigest(), + ) + frozen=Path(workspace["path"]) + self.assertEqual((frozen/"src/app.py").read_text(),"VALUE = 'candidate'\n") + self.assertEqual( + subprocess.run( + ["git","-C",str(frozen),"rev-parse","--abbrev-ref","HEAD"], + check=True,text=True,capture_output=True, + ).stdout.strip(), + "HEAD", + ) + self.assertEqual( + ( + subprocess.run( + ["git","-C",str(self.repo),"rev-parse","HEAD"], + check=True,capture_output=True, + ).stdout, + subprocess.run( + ["git","-C",str(self.repo),"status","--porcelain=v1","-z"], + check=True,capture_output=True, + ).stdout, + hashlib.sha256((self.repo/".git/index").read_bytes()).hexdigest(), + ), + (before_head,before_status,before_index), + ) (self.repo/"profiles.json").write_text("not valid after snapshot\n") (self.repo/"policy.json").write_text("also changed\n") @@ -534,6 +601,65 @@ def test_public_preflight_freezes_profile_policy_and_alias_selection(self): finally: store.close() + def test_public_preflight_reads_config_from_frozen_target_commit(self): + frozen_target=subprocess.run( + ["git","-C",str(self.repo),"rev-parse","HEAD"], + check=True,text=True,capture_output=True, + ).stdout.strip() + (self.repo/"profiles.json").write_text("invalid current branch profile\n") + (self.repo/"policy.json").write_text("invalid current branch policy\n") + subprocess.run(["git","-C",str(self.repo),"add","profiles.json","policy.json"],check=True) + subprocess.run( + ["git","-C",str(self.repo),"commit","-qm","move current configs"], + check=True, + ) + task=json.loads(json.dumps(self.task)) + task["project"]["base_ref"]=frozen_target + task["project"]["target_ref"]=frozen_target + + started=self.service.start(task,"target-config") + self.assertEqual(started["error"]["error"],"CAPABILITY_UNAVAILABLE") + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + snapshot=json.loads(store.run(started["run_id"])["mutable_snapshot"]) + finally: + store.close() + self.assertEqual(snapshot["target_oid"],frozen_target) + self.assertEqual( + snapshot["configs"]["profiles_file"]["sha256"], + hashlib.sha256(self.profiles_json.encode()).hexdigest(), + ) + self.assertEqual( + snapshot["configs"]["policy_file"]["sha256"], + hashlib.sha256(self.policy_json.encode()).hexdigest(), + ) + self.assertEqual( + subprocess.run( + ["git","-C",snapshot["workspace"]["path"],"rev-parse","HEAD"], + check=True,text=True,capture_output=True, + ).stdout.strip(), + frozen_target, + ) + + def test_public_preflight_rejects_dirty_scoped_input_before_workspace(self): + (self.repo/"src/uncommitted.py").write_text("dirty\n") + started=self.service.start(self.task,"dirty-scope") + self.assertEqual(started["error"]["error"],"PREPARATION_FAILED") + self.assertIn("src/uncommitted.py",started["error"]["message"]) + store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") + try: + run=store.run(started["run_id"]) + self.assertIsNone(run["mutable_snapshot"]) + self.assertIsNotNone(store.artifact_named( + started["run_id"],"result-receipt.json", + )) + finally: + store.close() + self.assertFalse( + (self.runtime/"projects"/run["project_id"]/"runs"/ + started["run_id"]/"review-worktree").exists() + ) + def test_invalid_predecessor_fails_with_run_context_and_receipt(self): started=self.service.start( self.task,"invalid-predecessor","does-not-exist",_internal_fake_delay=.01, diff --git a/test/core/test_store.py b/test/core/test_store.py index d16d1b8..c2e06ca 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -11,6 +11,7 @@ import sys sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.contracts import ContractError from devsquad.store import ConflictError, SchemaVersionError, Store, git_common_dir @@ -160,6 +161,62 @@ def test_git_common_dir_unifies_linked_worktrees(self): b = self.store.claim_start(linked, "linked", {"task": 2}, "b") self.assertEqual(a.project_id, b.project_id) + def test_preparation_can_pin_a_same_project_frozen_worktree(self): + frozen = self.root / "frozen-review" + subprocess.run( + ["git", "-C", str(self.repo), "worktree", "add", "--detach", "-q", + str(frozen), "HEAD"], + check=True, + ) + target_oid = subprocess.run( + ["git", "-C", str(self.repo), "rev-parse", "HEAD"], + check=True, + text=True, + capture_output=True, + ).stdout.strip() + claim = self.store.claim_start(self.repo, "frozen", {"task": 1}, "owner") + version = self.store.complete_preparation( + claim.run_id, + claim.fencing_token, + {"target_oid": target_oid}, + package_path="/frozen/package", + package_digest="package", + worktree_path=str(frozen), + ) + self.assertEqual(self.store.run(claim.run_id)["worktree_path"], str(frozen.resolve())) + reservation = self.store.reserve_attempt( + claim.run_id, version, "supervisor", "package", + ) + attempt = self.store.connection.execute( + "SELECT worktree_path FROM attempts WHERE id=?", (reservation.attempt_id,), + ).fetchone() + self.assertEqual(attempt["worktree_path"], str(frozen.resolve())) + + other = self.root / "other" + subprocess.run(["git", "init", "-q", str(other)], check=True) + subprocess.run( + ["git", "-C", str(other), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(other), "config", "user.name", "Test"], check=True, + ) + (other / "README").write_text("other\n") + subprocess.run(["git", "-C", str(other), "add", "README"], check=True) + subprocess.run(["git", "-C", str(other), "commit", "-qm", "other"], check=True) + rejected = self.store.claim_start(self.repo, "wrong-project", {"task": 2}, "owner") + with self.assertRaisesRegex(ContractError, "different project"): + self.store.complete_preparation( + rejected.run_id, + rejected.fencing_token, + {"target_oid": target_oid}, + worktree_path=str(other), + ) + self.assertEqual( + (self.store.run(rejected.run_id)["state"], self.store.run(rejected.run_id)["phase"]), + ("queued", "preparing"), + ) + def test_artifact_is_finalized_and_verified_before_reference(self): claim = self.store.claim_start(self.repo, "artifact", {"task": 1}, "owner") path, digest, size = self.store.finalize_artifact(claim.run_id, "result.json", b'{"ok":true}') diff --git a/test/core/test_workspaces.py b/test/core/test_workspaces.py new file mode 100644 index 0000000..ae5f889 --- /dev/null +++ b/test/core/test_workspaces.py @@ -0,0 +1,190 @@ +import hashlib +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.workspaces import ( + assert_clean_inputs, + committed_regular_file, + prepare_review_workspace, + repo_relative_config, + resolve_commit, +) + + +class ReviewWorkspaceTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-workspace-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], check=True, + ) + for directory in ("src", "tests", "devsquad"): + (self.repo / directory).mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# base test\n") + (self.repo / "devsquad/profiles.json").write_text('{"profiles":"base"}\n') + (self.repo / "devsquad/policy.json").write_text('{"policy":"base"}\n') + subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.base = resolve_commit(self.repo, "HEAD") + (self.repo / "src/app.py").write_text("VALUE = 'candidate'\n") + subprocess.run(["git", "-C", str(self.repo), "add", "src/app.py"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "candidate"], check=True, + ) + self.target = resolve_commit(self.repo, "HEAD") + + def prepare(self, run_id="run-1", target=None, scope=("src", "tests")): + return prepare_review_workspace( + self.repo, + self.runtime, + "project-1", + run_id, + self.base, + target or self.target, + scope, + required_clean_paths=("devsquad/profiles.json", "devsquad/policy.json"), + ) + + def git(self, *args): + return subprocess.run( + ["git", "-C", str(self.repo), *args], + check=True, + stdout=subprocess.PIPE, + ).stdout + + def test_workspace_is_detached_frozen_idempotent_and_checkout_preserving(self): + (self.repo / "notes.txt").write_text("dirty but outside declared inputs\n") + before_head = self.git("rev-parse", "HEAD") + before_status = self.git("status", "--porcelain=v1", "-z") + before_index = hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest() + + first = self.prepare() + after_head = self.git("rev-parse", "HEAD") + after_status = self.git("status", "--porcelain=v1", "-z") + after_index = hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest() + self.assertEqual((after_head, after_status, after_index), ( + before_head, before_status, before_index, + )) + workspace = Path(first["path"]) + self.assertEqual((workspace / "src/app.py").read_text(), "VALUE = 'candidate'\n") + self.assertEqual( + subprocess.run( + ["git", "-C", str(workspace), "rev-parse", "--abbrev-ref", "HEAD"], + check=True, + text=True, + capture_output=True, + ).stdout.strip(), + "HEAD", + ) + self.assertEqual(first["base_oid"], self.base) + self.assertEqual(first["target_oid"], self.target) + self.assertEqual(first["changed_paths"], ["src/app.py"]) + self.assertEqual(first["scope"], ["src", "tests"]) + + (self.repo / "src/app.py").write_text("VALUE = 'later'\n") + subprocess.run(["git", "-C", str(self.repo), "add", "src/app.py"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "later"], check=True) + second = self.prepare() + self.assertEqual(second, first) + self.assertEqual((workspace / "src/app.py").read_text(), "VALUE = 'candidate'\n") + + def test_dirty_scoped_tracked_path_is_rejected(self): + (self.repo / "src/app.py").write_text("dirty\n") + with self.assertRaisesRegex(ContractError, "src/app.py"): + self.prepare() + + def test_dirty_required_config_is_rejected_even_outside_scope(self): + (self.repo / "devsquad/policy.json").write_text("dirty\n") + subprocess.run( + ["git", "-C", str(self.repo), "add", "devsquad/policy.json"], check=True, + ) + with self.assertRaisesRegex(ContractError, "devsquad/policy.json"): + self.prepare(scope=("src",)) + + def test_untracked_scoped_path_is_rejected(self): + (self.repo / "tests/new_test.py").write_text("dirty\n") + with self.assertRaisesRegex(ContractError, "tests/new_test.py"): + self.prepare() + + def test_scoped_symlink_escape_is_rejected(self): + outside = self.root / "outside.txt" + outside.write_text("private\n") + (self.repo / "src/leak").symlink_to(outside) + subprocess.run(["git", "-C", str(self.repo), "add", "src/leak"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "symlink"], check=True) + target = resolve_commit(self.repo, "HEAD") + with self.assertRaisesRegex(ContractError, "symlink escapes"): + self.prepare(run_id="symlink-run", target=target) + + def test_existing_workspace_for_another_target_is_rejected(self): + self.prepare(run_id="reused-path") + with self.assertRaisesRegex(ContractError, "different commit"): + self.prepare(run_id="reused-path", target=self.base) + + def test_committed_config_reads_exact_regular_blob(self): + expected = (self.repo / "devsquad/policy.json").read_bytes() + self.assertEqual( + committed_regular_file(self.repo, self.target, "devsquad/policy.json"), + expected, + ) + outside = self.root / "outside-config" + outside.write_text("outside\n") + (self.repo / "devsquad/link.json").symlink_to(outside) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad/link.json"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "linked config"], check=True, + ) + with self.assertRaisesRegex(ContractError, "regular file"): + committed_regular_file( + self.repo, resolve_commit(self.repo, "HEAD"), "devsquad/link.json", + ) + + def test_config_paths_and_identifiers_cannot_escape(self): + self.assertEqual( + repo_relative_config( + self.repo, str(self.repo / "devsquad/policy.json"), "policy_file", + ), + "devsquad/policy.json", + ) + with self.assertRaisesRegex(ContractError, "without traversal"): + repo_relative_config(self.repo, "../policy.json", "policy_file") + with self.assertRaisesRegex(ContractError, "escapes project"): + repo_relative_config(self.repo, str(self.root / "outside.json"), "policy_file") + with self.assertRaisesRegex(ContractError, "safe path segment"): + prepare_review_workspace( + self.repo, + self.runtime, + "../project", + "run", + self.base, + self.target, + ("src",), + ) + + def test_clean_input_helper_accepts_unrelated_dirty_files(self): + (self.repo / "notes.txt").write_text("unrelated\n") + assert_clean_inputs( + self.repo, + ("src", "tests"), + ("devsquad/profiles.json", "devsquad/policy.json"), + ) + + +if __name__ == "__main__": + unittest.main() From 4b5f308beb7db08dd52f09f8b8b0144288bbaf92 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 16 Sep 2026 01:13:06 +0530 Subject: [PATCH 050/197] docs: checkpoint M3 frozen preflight --- docs/plans/engineering-team/RESUME.md | 44 ++++++++++++++++-------- docs/plans/engineering-team/backlog.json | 32 +++++++++++++++-- 2 files changed, 59 insertions(+), 17 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index b7310d7..abe0c69 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -22,9 +22,19 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne bounded target. `ddb6f51` fixes both: lease authorization now samples time after acquiring the SQLite write transaction, and public cancel resumes an interrupted `recovery_cleanup`. Both have deterministic regressions. -- The next milestone is M3. Public `branch-review` still fails explicitly as - `CAPABILITY_UNAVAILABLE`; M2 is infrastructure, not yet a usable engineering - workflow. Do not advertise DevSquad as ready for real review until M3 passes. +- M3 is in progress through `cd9a881`. `ca55990` adds strict deterministic + profile/policy routing, alias binding snapshots, pins/fallbacks, independent + reviewer selection and typed pool capacity. `1c7b614` freezes those exact + bytes and decisions during public preflight. `cd9a881` resolves base/target + OIDs and creates a detached run-owned review worktree without changing the + submitted checkout, index or HEAD; dirty scoped/config inputs and escaping + symlinks fail closed. The current gate is 144 core tests with + `ResourceWarning` promoted to failure plus 202 Bash assertions. +- Public `branch-review` still fails explicitly as `CAPABILITY_UNAVAILABLE` + after the now-tested M3 preflight. The reviewer, declared checks, lead + disposition and bound report/receipt pipeline are not implemented yet. Do + not advertise DevSquad as ready for real review until that end-to-end gate + passes. - Current provider readiness is external to M2: Claude CLI is not logged in; Grok CLI authentication expired; Gemini CLI's individual-account path is unsupported and its supported successor is Antigravity; Antigravity is @@ -34,7 +44,9 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne completes the corresponding normal login/permission action. - User wants the implementation orchestrated efficiently and preserved across Plus-plan interruptions. Avoid recursive subagent fan-out: it consumed the - shared window rapidly without advancing M3. + shared window rapidly without advancing M3. The recursively created M3 + planning agents all hit the same Plus limit; continue locally until shared + agent capacity is restored, then use only bounded leaf reviews. - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M3 is next. @@ -48,12 +60,14 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 121 tests passed at the M2 acceptance gate | +| Python core discovery | 144 tests passed through the M3 frozen-workspace checkpoint | | Bash 3.2 regression suite | 10 test files, 202 assertions passed | | Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 5 | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | | Integrated native adapter proof | Passed at `97a10f0`; gpt-5.5/low, read-only, correlated completion and confirmed process-group cleanup | | M2 crash/race matrix | Real subprocess interruptions plus independent-process start, writer, cancel, import and host-handoff races passed at `ddb6f51` | +| M3 deterministic routing | Strict profiles/policy, aliases, overrides, fallback and typed capacity tests passed at `ca55990` / `1c7b614` | +| M3 frozen review input | Exact OIDs/config hashes, detached worktree, moving-ref stability, dirty-input rejection and source checkout preservation passed at `cd9a881` | The first two saved-probe invocations failed before `Popen` because of probe-only path/field defects, so neither launched Codex nor consumed a model @@ -73,16 +87,16 @@ are not advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Read the M3 section of [IMPLEMENTATION.md](IMPLEMENTATION.md), the task, - Profile/Policy and workflow sections of [CONTRACTS.md](CONTRACTS.md), and - the selection amendment. Implement the earliest M3 dependency: strict - profile/policy loading plus deterministic frozen role bindings and fallback - decisions, with tests. Then add frozen review workspace preparation before - any real reviewer launch. -3. Preserve the M2 process/fencing boundaries. Extend the durable runner with - real workflow steps; do not bypass it with an in-process or ad-hoc provider - call. Keep public review unavailable until reviewer, declared checks, lead - disposition and bound receipt artifacts form a valid end-to-end slice. +2. Preserve the M2 process/fencing boundaries and implement the next M3 slice: + the durable reviewer → trusted declared checks → lead disposition pipeline. + Add strict structured review/check/disposition artifacts bound to + `workspace.candidate_sha256`, explicit `required_to_pass` handling and host + handoff continuation. Do not bypass the supervisor with an in-process or + ad-hoc provider call. +3. Add terminal `receipt.json`, `receipt.md`, `events.jsonl` and artifact + manifest generation for the workflow. Keep public review unavailable until + the complete fake-adapter end-to-end gate passes, then run one bounded real + Codex review before declaring the M3 product stop usable. 4. Use offline fake adapters for development. Provider login/permission work is a later live gate and must not block independent M3 implementation. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 57c8c14..a2d8f9d 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -82,9 +82,37 @@ "id": "M3", "title": "Usable branch review from terminal", "depends_on": ["M2"], - "status": "pending", + "status": "in_progress", "acceptance_section": "M3 — Ship a useful branch review", - "evidence": [], + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "ca55990", + "command_or_action": "131 core tests and 202 shell assertions covering strict deterministic routing, versioned aliases, pins/fallbacks, independent review and typed pool capacity", + "outcome": "M3 routing decisions are deterministic, fail closed and retain exact profile/policy identity; workflow execution remains unavailable", + "artifact": "../../../test/core/test_router.py", + "recorded_at": "2026-09-16T00:52:00+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "1c7b614", + "command_or_action": "132 core tests and 202 shell assertions covering public-preflight routing snapshots and replay after mutable config corruption", + "outcome": "New runs freeze exact config hashes, bindings and selected concrete profiles before execution", + "artifact": "../../../test/core/test_service.py", + "recorded_at": "2026-09-16T00:58:00+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "cd9a881", + "command_or_action": "144 core tests with ResourceWarning promoted to error and 202 shell assertions covering exact commit inputs and run-owned detached worktrees", + "outcome": "M3 preflight freezes base/target/config identity, rejects dirty scoped inputs and escaping symlinks, and preserves the submitted checkout/index/HEAD; reviewer/check/lead execution remains pending", + "artifact": "../../../test/core/test_workspaces.py", + "recorded_at": "2026-09-16T01:08:00+05:30", + "availability": "tracked_tests" + } + ], "blocker": null }, { From a756307c0e5be49a30fe0b570c16f448b25479bc Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 16 Sep 2026 01:21:23 +0530 Subject: [PATCH 051/197] feat: define strict M3 workflow evidence --- plugin/core/schemas/check-result.schema.json | 36 ++ plugin/core/schemas/review-result.schema.json | 38 ++ plugin/core/src/devsquad/validation.py | 7 + plugin/core/src/devsquad/workflows.py | 394 ++++++++++++++++++ test/core/test_review_workflow.py | 248 +++++++++++ test/core/test_validation.py | 2 + 6 files changed, 725 insertions(+) create mode 100644 plugin/core/schemas/check-result.schema.json create mode 100644 plugin/core/schemas/review-result.schema.json create mode 100644 plugin/core/src/devsquad/workflows.py create mode 100644 test/core/test_review_workflow.py diff --git a/plugin/core/schemas/check-result.schema.json b/plugin/core/schemas/check-result.schema.json new file mode 100644 index 0000000..0e782e1 --- /dev/null +++ b/plugin/core/schemas/check-result.schema.json @@ -0,0 +1,36 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/check-result-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "candidate_sha256", "target_oid", "id", "argv", "cwd", "required_to_pass", "status", "returncode", "error_code", "duration_ms", "stdout", "stderr"], + "properties": { + "schema_version": {"const": 1}, + "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "target_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "id": {"type": "string", "minLength": 1}, + "argv": {"type": "array", "minItems": 1, "items": {"type": "string", "minLength": 1}}, + "cwd": {"type": "string"}, + "required_to_pass": {"type": "boolean"}, + "status": {"enum": ["passed", "failed", "timed_out", "launch_failed"]}, + "returncode": {"type": ["integer", "null"]}, + "error_code": {"enum": [null, "TIMEOUT", "CLI_ERROR"]}, + "duration_ms": {"type": "integer", "minimum": 0}, + "stdout": {"$ref": "#/$defs/stream"}, + "stderr": {"$ref": "#/$defs/stream"} + }, + "$defs": { + "stream": { + "type": "object", + "additionalProperties": false, + "required": ["preview", "captured_bytes", "total_bytes", "truncated", "full_sha256"], + "properties": { + "preview": {"type": "string", "maxLength": 65536}, + "captured_bytes": {"type": "integer", "minimum": 0}, + "total_bytes": {"type": "integer", "minimum": 0}, + "truncated": {"type": "boolean"}, + "full_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"} + } + } + } +} diff --git a/plugin/core/schemas/review-result.schema.json b/plugin/core/schemas/review-result.schema.json new file mode 100644 index 0000000..95e673d --- /dev/null +++ b/plugin/core/schemas/review-result.schema.json @@ -0,0 +1,38 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/review-result-v1.json", + "type": "object", + "additionalProperties": false, + "required": ["schema_version", "candidate_sha256", "base_oid", "target_oid", "review_mode", "verdict", "summary", "findings"], + "properties": { + "schema_version": {"const": 1}, + "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "base_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "target_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "review_mode": {"enum": ["standard", "adversarial"]}, + "verdict": {"enum": ["clean", "findings"]}, + "summary": {"type": "string", "minLength": 1, "maxLength": 20000}, + "findings": { + "type": "array", + "maxItems": 100, + "items": {"$ref": "#/$defs/finding"} + } + }, + "$defs": { + "finding": { + "type": "object", + "additionalProperties": false, + "required": ["id", "severity", "title", "description", "path", "start_line", "end_line", "evidence"], + "properties": { + "id": {"type": "string", "minLength": 1, "maxLength": 200}, + "severity": {"enum": ["critical", "high", "medium", "low"]}, + "title": {"type": "string", "minLength": 1, "maxLength": 500}, + "description": {"type": "string", "minLength": 1, "maxLength": 20000}, + "path": {"type": "string", "minLength": 1}, + "start_line": {"type": "integer", "minimum": 1}, + "end_line": {"type": "integer", "minimum": 1}, + "evidence": {"type": "string", "minLength": 1, "maxLength": 20000} + } + } + } +} diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py index 663af77..ee51702 100644 --- a/plugin/core/src/devsquad/validation.py +++ b/plugin/core/src/devsquad/validation.py @@ -36,16 +36,23 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) raise ContractError("goal and task_class must be non-empty strings") if not isinstance(value["acceptance"], list) or not value["acceptance"]: raise ContractError("acceptance must be non-empty") + acceptance_ids = set() for item in value["acceptance"]: _exact(item, {"id", "description", "evidence_kind"}, {"id", "description", "evidence_kind"}, "acceptance item") if not all(isinstance(item[k], str) and item[k].strip() for k in ("id", "description")): raise ContractError("acceptance id and description must be non-empty strings") if not isinstance(item["evidence_kind"], str) or item["evidence_kind"] not in {"review", "check", "artifact", "host"}: raise ContractError("invalid evidence_kind") + if item["id"] in acceptance_ids: + raise ContractError("acceptance ids must be unique") + acceptance_ids.add(item["id"]) if not isinstance(value["checks"], list): raise ContractError("checks must be an array") + check_ids = set() for check in value["checks"]: _exact(check, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, "check") if not isinstance(check["id"], str) or not check["id"]: raise ContractError("check id must be non-empty") + if check["id"] in check_ids: raise ContractError("check ids must be unique") + check_ids.add(check["id"]) if not isinstance(check["argv"], list) or not check["argv"] or not all(isinstance(v, str) and v for v in check["argv"]): raise ContractError("check argv must be a non-empty string array") _relative(check["cwd"], "check cwd") if not isinstance(check["timeout_seconds"], int) or isinstance(check["timeout_seconds"], bool) or check["timeout_seconds"] <= 0: raise ContractError("check timeout must be positive") diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py new file mode 100644 index 0000000..3e2b856 --- /dev/null +++ b/plugin/core/src/devsquad/workflows.py @@ -0,0 +1,394 @@ +"""Strict, side-effect-free contracts for the two fixed engineering workflows.""" + +from __future__ import annotations + +import json +from pathlib import PurePosixPath +import re +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .validation import validate_task + + +MAX_REVIEW_BYTES = 512 * 1024 +MAX_FINDINGS = 100 +MAX_TEXT_CHARS = 20_000 +MAX_PREVIEW_CHARS = 64 * 1024 +FINDING_SEVERITIES = {"critical", "high", "medium", "low"} +CHECK_STATUSES = {"passed", "failed", "timed_out", "launch_failed"} +REVIEW_MODES = {"standard", "adversarial"} +_SHA256 = re.compile(r"[0-9a-f]{64}\Z") +_COMMIT_OID = re.compile(r"[0-9a-f]{40}\Z") + + +def _exact( + value: Any, + fields: set[str], + label: str, +) -> dict[str, Any]: + if not isinstance(value, dict): + raise ContractError(f"{label} must be an object") + unknown, missing = set(value) - fields, fields - set(value) + if unknown or missing: + raise ContractError( + f"{label} fields invalid: unknown={sorted(unknown)} " + f"missing={sorted(missing)}" + ) + return value + + +def _text(value: Any, label: str, *, maximum: int = MAX_TEXT_CHARS) -> str: + if not isinstance(value, str) or not value.strip(): + raise ContractError(f"{label} must be a non-empty string") + if len(value) > maximum: + raise ContractError(f"{label} exceeds its size limit") + return value + + +def _sha256(value: Any, label: str) -> str: + if not isinstance(value, str) or _SHA256.fullmatch(value) is None: + raise ContractError(f"{label} must be a lowercase SHA-256") + return value + + +def _commit_oid(value: Any, label: str) -> str: + if not isinstance(value, str) or _COMMIT_OID.fullmatch(value) is None: + raise ContractError(f"{label} must be a full lowercase commit OID") + return value + + +def _relative_path(value: Any, label: str) -> str: + if (not isinstance(value, str) or not value or "\\" in value + or "\0" in value): + raise ContractError(f"{label} must be a canonical repository-relative path") + path = PurePosixPath(value) + normalized = path.as_posix() + if (path.is_absolute() or ".." in path.parts or normalized in {"", "."} + or normalized != value): + raise ContractError(f"{label} must be a canonical repository-relative path") + return normalized + + +def _inside_scope(path: str, scopes: list[str]) -> bool: + for raw_scope in scopes: + scope = PurePosixPath(raw_scope).as_posix().rstrip("/") or "." + if scope == "." or path == scope or path.startswith(f"{scope}/"): + return True + return False + + +def _strict_json_object(payload: bytes | str, label: str) -> dict[str, Any]: + if isinstance(payload, str): + encoded = payload.encode() + elif isinstance(payload, bytes): + encoded = payload + else: + raise ContractError(f"{label} must be UTF-8 JSON bytes or text") + if len(encoded) > MAX_REVIEW_BYTES: + raise ContractError(f"{label} exceeds its byte limit") + + def object_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result = {} + for key, value in pairs: + if key in result: + raise ContractError(f"{label} contains duplicate key: {key}") + result[key] = value + return result + + def reject_constant(value: str) -> None: + raise ContractError(f"{label} contains non-finite number: {value}") + + try: + value = json.loads( + encoded.decode("utf-8"), + object_pairs_hook=object_pairs, + parse_constant=reject_constant, + ) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError(f"{label} is not valid UTF-8 JSON") from exc + if not isinstance(value, dict): + raise ContractError(f"{label} must contain one JSON object") + return value + + +def _workspace_identity(workspace: dict[str, Any]) -> tuple[str, str, str]: + if not isinstance(workspace, dict): + raise ContractError("workspace snapshot must be an object") + return ( + _sha256(workspace.get("candidate_sha256"), "workspace candidate_sha256"), + _commit_oid(workspace.get("base_oid"), "workspace base_oid"), + _commit_oid(workspace.get("target_oid"), "workspace target_oid"), + ) + + +def review_mode(task: dict[str, Any]) -> str: + mode = task.get("review", {}).get("mode", "standard") + if mode not in REVIEW_MODES: + raise ContractError("review mode is invalid") + return mode + + +def validate_review_document( + value: dict[str, Any], + task: dict[str, Any], + workspace: dict[str, Any], +) -> dict[str, Any]: + """Validate model output and bind it to the exact frozen candidate.""" + validate_task(task) + if task["workflow"] != "branch-review": + raise ContractError("review document requires a branch-review task") + document = _exact(value, { + "schema_version", "candidate_sha256", "base_oid", "target_oid", + "review_mode", "verdict", "summary", "findings", + }, "review document") + if document["schema_version"] != 1 or type(document["schema_version"]) is not int: + raise ContractError("review document schema_version is invalid") + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + if _sha256(document["candidate_sha256"], "review candidate_sha256") != candidate_sha256: + raise ContractError("review targets a different candidate hash") + if _commit_oid(document["base_oid"], "review base_oid") != base_oid: + raise ContractError("review targets a different base commit") + if _commit_oid(document["target_oid"], "review target_oid") != target_oid: + raise ContractError("review targets a different target commit") + if document["review_mode"] != review_mode(task): + raise ContractError("review mode does not match the frozen task") + if document["verdict"] not in {"clean", "findings"}: + raise ContractError("review verdict must be clean or findings") + _text(document["summary"], "review summary") + findings = document["findings"] + if not isinstance(findings, list) or len(findings) > MAX_FINDINGS: + raise ContractError("review findings must be a bounded array") + seen = set() + for finding in findings: + item = _exact(finding, { + "id", "severity", "title", "description", "path", + "start_line", "end_line", "evidence", + }, "review finding") + finding_id = _text(item["id"], "finding id", maximum=200) + if finding_id in seen: + raise ContractError(f"review finding id is duplicated: {finding_id}") + seen.add(finding_id) + if item["severity"] not in FINDING_SEVERITIES: + raise ContractError("review finding severity is invalid") + _text(item["title"], "finding title", maximum=500) + _text(item["description"], "finding description") + finding_path = _relative_path(item["path"], "finding path") + if not _inside_scope(finding_path, task["scope"]["read_paths"]): + raise ContractError("review finding path is outside the declared read scope") + if type(item["start_line"]) is not int or item["start_line"] < 1: + raise ContractError("finding start_line must be a positive integer") + if type(item["end_line"]) is not int or item["end_line"] < item["start_line"]: + raise ContractError("finding end_line must not precede start_line") + _text(item["evidence"], "finding evidence") + if (document["verdict"] == "clean") != (not findings): + raise ContractError("review verdict and findings disagree") + return json.loads(canonical_json(document)) + + +def decode_review_document( + payload: bytes | str, + task: dict[str, Any], + workspace: dict[str, Any], +) -> dict[str, Any]: + return validate_review_document( + _strict_json_object(payload, "review output"), task, workspace, + ) + + +def _validate_stream(value: Any, label: str) -> None: + stream = _exact(value, { + "preview", "captured_bytes", "total_bytes", "truncated", "full_sha256", + }, label) + if not isinstance(stream["preview"], str) or len(stream["preview"]) > MAX_PREVIEW_CHARS: + raise ContractError(f"{label} preview exceeds its size limit") + for field in ("captured_bytes", "total_bytes"): + if type(stream[field]) is not int or stream[field] < 0: + raise ContractError(f"{label} {field} must be a non-negative integer") + if stream["captured_bytes"] > stream["total_bytes"]: + raise ContractError(f"{label} captured bytes exceed total bytes") + if type(stream["truncated"]) is not bool: + raise ContractError(f"{label} truncated must be boolean") + if stream["truncated"] != (stream["captured_bytes"] < stream["total_bytes"]): + raise ContractError(f"{label} truncation metadata is inconsistent") + _sha256(stream["full_sha256"], f"{label} full_sha256") + + +def validate_check_results( + values: list[dict[str, Any]], + task: dict[str, Any], + workspace: dict[str, Any], +) -> list[dict[str, Any]]: + """Validate check evidence against the host-supplied immutable check plan.""" + validate_task(task) + if not isinstance(values, list) or len(values) != len(task["checks"]): + raise ContractError("check results do not match the declared check count") + candidate_sha256, _, target_oid = _workspace_identity(workspace) + normalized = [] + for configured, supplied in zip(task["checks"], values): + result = _exact(supplied, { + "schema_version", "candidate_sha256", "target_oid", "id", "argv", + "cwd", "required_to_pass", "status", "returncode", "error_code", + "duration_ms", "stdout", "stderr", + }, "check result") + if result["schema_version"] != 1 or type(result["schema_version"]) is not int: + raise ContractError("check result schema_version is invalid") + if _sha256(result["candidate_sha256"], "check candidate_sha256") != candidate_sha256: + raise ContractError("check result targets a different candidate hash") + if _commit_oid(result["target_oid"], "check target_oid") != target_oid: + raise ContractError("check result targets a different target commit") + for field in ("id", "argv", "cwd", "required_to_pass"): + if result[field] != configured[field]: + raise ContractError(f"check result changes declared field: {field}") + if result["status"] not in CHECK_STATUSES: + raise ContractError("check result status is invalid") + if type(result["duration_ms"]) is not int or result["duration_ms"] < 0: + raise ContractError("check duration_ms must be a non-negative integer") + status, returncode, error_code = ( + result["status"], result["returncode"], result["error_code"] + ) + if status == "passed" and (returncode != 0 or error_code is not None): + raise ContractError("passing check result is inconsistent") + if status == "failed" and ( + type(returncode) is not int or returncode == 0 or error_code is not None + ): + raise ContractError("failed check result is inconsistent") + if status == "timed_out" and error_code != "TIMEOUT": + raise ContractError("timed-out check result is inconsistent") + if status == "launch_failed" and ( + returncode is not None or error_code != "CLI_ERROR" + ): + raise ContractError("launch-failed check result is inconsistent") + if status in {"timed_out", "launch_failed"} and returncode is not None: + raise ContractError("incomplete check cannot report a return code") + _validate_stream(result["stdout"], "check stdout") + _validate_stream(result["stderr"], "check stderr") + normalized.append(json.loads(canonical_json(result))) + return normalized + + +def evaluate_branch_review( + task: dict[str, Any], + workspace: dict[str, Any], + review: dict[str, Any], + checks: list[dict[str, Any]], +) -> dict[str, Any]: + """Derive non-overridable gates without pretending to make the lead decision.""" + normalized_review = validate_review_document(review, task, workspace) + normalized_checks = validate_check_results(checks, task, workspace) + required_failures = [ + result["id"] for result in normalized_checks + if result["required_to_pass"] and result["status"] != "passed" + ] + report_only_failures = [ + result["id"] for result in normalized_checks + if not result["required_to_pass"] and result["status"] != "passed" + ] + criteria = [] + for criterion in task["acceptance"]: + kind = criterion["evidence_kind"] + if kind == "review": + status, evidence = "evidence_available", ["review.json"] + elif kind == "check" and normalized_checks: + status = "evidence_available" + evidence = [f"check:{result['id']}" for result in normalized_checks] + else: + status, evidence = "pending_lead", [] + criteria.append({ + "id": criterion["id"], + "evidence_kind": kind, + "status": status, + "evidence": evidence, + }) + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + return { + "schema_version": 1, + "candidate_sha256": candidate_sha256, + "base_oid": base_oid, + "target_oid": target_oid, + "review_verdict": normalized_review["verdict"], + "required_checks_passed": not required_failures, + "required_failures": required_failures, + "report_only_failures": report_only_failures, + "accept_allowed": not required_failures, + "accept_blockers": [ + f"required_check_failed:{check_id}" for check_id in required_failures + ], + "criteria": criteria, + } + + +def apply_lead_disposition( + evaluation: dict[str, Any], + disposition: str, + *, + revisions_used: int, + max_revisions: int, +) -> dict[str, Any]: + """Apply the fixed lead gate without allowing prose to bypass evidence.""" + if disposition not in {"accept", "revise", "reject"}: + raise ContractError("lead disposition is invalid") + if (type(revisions_used) is not int or revisions_used < 0 + or type(max_revisions) is not int or max_revisions < 0): + raise ContractError("revision counters must be non-negative integers") + if disposition == "accept": + if evaluation.get("accept_allowed") is not True: + raise ContractError("lead acceptance is blocked by required evidence") + action, terminal_state = "complete", "succeeded" + elif disposition == "reject": + action, terminal_state = "complete", "failed" + elif revisions_used >= max_revisions: + action, terminal_state = "budget_exhausted", "failed" + else: + action, terminal_state = "repeat_review", None + return { + "disposition": disposition, + "action": action, + "terminal_state": terminal_state, + "next_revision": revisions_used + 1 if action == "repeat_review" else revisions_used, + } + + +def build_review_prompt(task: dict[str, Any], workspace: dict[str, Any]) -> str: + """Build a deterministic, read-only prompt from host-authorized fields only.""" + validate_task(task) + if task["workflow"] != "branch-review": + raise ContractError("review prompt requires a branch-review task") + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + mode = review_mode(task) + focus = task.get("review", {}).get("focus") + assignment = { + "goal": task["goal"], + "acceptance": task["acceptance"], + "scope": task["scope"]["read_paths"], + "review_mode": mode, + "focus": focus, + "base_oid": base_oid, + "target_oid": target_oid, + "candidate_sha256": candidate_sha256, + } + finding_shape = { + "id": "finding-id", + "severity": "critical|high|medium|low", + "title": "short title", + "description": "actionable explanation", + "path": "repository/relative/path", + "start_line": 1, + "end_line": 1, + "evidence": "specific supporting evidence", + } + return "\n".join([ + "You are the read-only reviewer for one frozen Git candidate.", + "Do not edit files, run mutating commands, publish, delegate, or broaden scope.", + "Inspect only the declared scope and compare the exact base and target commits.", + "A valid critical finding is useful output; do not hide findings to claim success.", + "Return exactly one JSON object and no Markdown or surrounding prose.", + "The object must contain these exact top-level fields:", + "schema_version, candidate_sha256, base_oid, target_oid, review_mode, verdict, summary, findings.", + "verdict must be clean with an empty findings array, or findings with at least one finding.", + "Each finding must have exactly this shape:", + canonical_json(finding_shape), + "Frozen assignment:", + canonical_json(assignment), + ]) diff --git a/test/core/test_review_workflow.py b/test/core/test_review_workflow.py new file mode 100644 index 0000000..b35bacd --- /dev/null +++ b/test/core/test_review_workflow.py @@ -0,0 +1,248 @@ +import copy +import hashlib +import json +from pathlib import Path +import sys +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.workflows import ( + apply_lead_disposition, + build_review_prompt, + decode_review_document, + evaluate_branch_review, + validate_check_results, + validate_review_document, +) + + +class BranchReviewWorkflowTest(unittest.TestCase): + def setUp(self): + self.task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + self.task["project"]["repo_path"] = "/tmp/fixture-repository" + self.workspace = { + "candidate_sha256": "c" * 64, + "base_oid": "a" * 40, + "target_oid": "b" * 40, + } + self.review = { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "base_oid": "a" * 40, + "target_oid": "b" * 40, + "review_mode": "standard", + "verdict": "findings", + "summary": "One supported defect was found.", + "findings": [{ + "id": "F-1", + "severity": "high", + "title": "Incorrect boundary", + "description": "The candidate accepts one value beyond the limit.", + "path": "src/example.py", + "start_line": 12, + "end_line": 13, + "evidence": "The comparison uses <= where the contract requires <.", + }], + } + self.check = self.check_result("failed", returncode=1) + + def stream(self, content=""): + encoded = content.encode() + return { + "preview": content, + "captured_bytes": len(encoded), + "total_bytes": len(encoded), + "truncated": False, + "full_sha256": hashlib.sha256(encoded).hexdigest(), + } + + def check_result(self, status, *, returncode=None, error_code=None): + configured = self.task["checks"][0] + return { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "target_oid": "b" * 40, + "id": configured["id"], + "argv": configured["argv"], + "cwd": configured["cwd"], + "required_to_pass": configured["required_to_pass"], + "status": status, + "returncode": returncode, + "error_code": error_code, + "duration_ms": 12, + "stdout": self.stream("check output\n"), + "stderr": self.stream(), + } + + def test_review_is_strict_and_bound_to_candidate_commits_and_mode(self): + normalized = validate_review_document(self.review, self.task, self.workspace) + self.assertEqual(normalized, self.review) + for field, replacement, message in ( + ("candidate_sha256", "d" * 64, "different candidate"), + ("base_oid", "e" * 40, "different base"), + ("target_oid", "f" * 40, "different target"), + ("review_mode", "adversarial", "mode does not match"), + ): + invalid = copy.deepcopy(self.review) + invalid[field] = replacement + with self.assertRaisesRegex(ContractError, message): + validate_review_document(invalid, self.task, self.workspace) + + def test_review_rejects_unknown_duplicate_nonfinite_and_oversized_output(self): + unknown = copy.deepcopy(self.review) + unknown["extra"] = True + with self.assertRaisesRegex(ContractError, "fields invalid"): + validate_review_document(unknown, self.task, self.workspace) + duplicate = json.dumps(self.review)[:-1] + ',"summary":"replacement"}' + with self.assertRaisesRegex(ContractError, "duplicate key"): + decode_review_document(duplicate, self.task, self.workspace) + nonfinite = json.dumps(self.review).replace('"schema_version": 1', '"schema_version": NaN') + with self.assertRaisesRegex(ContractError, "non-finite"): + decode_review_document(nonfinite, self.task, self.workspace) + with self.assertRaisesRegex(ContractError, "byte limit"): + decode_review_document(b" " * (512 * 1024 + 1), self.task, self.workspace) + + def test_clean_and_findings_verdicts_cannot_contradict_payload(self): + clean = copy.deepcopy(self.review) + clean["verdict"], clean["findings"] = "clean", [] + self.assertEqual( + decode_review_document(json.dumps(clean), self.task, self.workspace), clean, + ) + for verdict, findings in (("clean", self.review["findings"]), ("findings", [])): + invalid = copy.deepcopy(self.review) + invalid["verdict"], invalid["findings"] = verdict, findings + with self.assertRaisesRegex(ContractError, "verdict and findings disagree"): + validate_review_document(invalid, self.task, self.workspace) + + def test_findings_require_actionable_canonical_locations_and_unique_ids(self): + for field, value in ( + ("path", "../outside.py"), + ("start_line", 0), + ("end_line", 11), + ("severity", "urgent"), + ): + invalid = copy.deepcopy(self.review) + invalid["findings"][0][field] = value + with self.assertRaises(ContractError): + validate_review_document(invalid, self.task, self.workspace) + outside_scope = copy.deepcopy(self.review) + outside_scope["findings"][0]["path"] = "docs/unreviewed.md" + with self.assertRaisesRegex(ContractError, "outside the declared read scope"): + validate_review_document(outside_scope, self.task, self.workspace) + duplicate = copy.deepcopy(self.review) + duplicate["findings"].append(copy.deepcopy(duplicate["findings"][0])) + with self.assertRaisesRegex(ContractError, "duplicated"): + validate_review_document(duplicate, self.task, self.workspace) + + def test_check_results_cannot_change_host_supplied_commands_or_candidate(self): + self.assertEqual( + validate_check_results([self.check], self.task, self.workspace), [self.check], + ) + for field, value in ( + ("id", "other"), + ("argv", ["sh", "-c", "echo widened"]), + ("cwd", "src"), + ("required_to_pass", True), + ): + invalid = copy.deepcopy(self.check) + invalid[field] = value + with self.assertRaisesRegex(ContractError, "changes declared field"): + validate_check_results([invalid], self.task, self.workspace) + invalid = copy.deepcopy(self.check) + invalid["candidate_sha256"] = "d" * 64 + with self.assertRaisesRegex(ContractError, "different candidate"): + validate_check_results([invalid], self.task, self.workspace) + + def test_check_status_and_bounded_stream_metadata_are_consistent(self): + passing = self.check_result("passed", returncode=0) + timed_out = self.check_result("timed_out", error_code="TIMEOUT") + launch_failed = self.check_result("launch_failed", error_code="CLI_ERROR") + for result in (passing, self.check, timed_out, launch_failed): + self.assertEqual( + validate_check_results([result], self.task, self.workspace), [result], + ) + invalid = copy.deepcopy(passing) + invalid["returncode"] = 1 + with self.assertRaisesRegex(ContractError, "passing check"): + validate_check_results([invalid], self.task, self.workspace) + invalid = copy.deepcopy(self.check) + invalid["stdout"]["truncated"] = True + with self.assertRaisesRegex(ContractError, "truncation metadata"): + validate_check_results([invalid], self.task, self.workspace) + + def test_report_only_failure_is_visible_but_does_not_block_delivery(self): + evaluation = evaluate_branch_review( + self.task, self.workspace, self.review, [self.check], + ) + self.assertTrue(evaluation["accept_allowed"]) + self.assertEqual(evaluation["required_failures"], []) + self.assertEqual(evaluation["report_only_failures"], ["fixture-tests"]) + self.assertEqual( + [item["status"] for item in evaluation["criteria"]], + ["evidence_available", "evidence_available"], + ) + accepted = apply_lead_disposition( + evaluation, "accept", revisions_used=0, max_revisions=0, + ) + self.assertEqual( + (accepted["action"], accepted["terminal_state"]), + ("complete", "succeeded"), + ) + + def test_required_failure_cannot_be_overridden_by_lead_prose(self): + required = copy.deepcopy(self.check) + required["required_to_pass"] = True + task = copy.deepcopy(self.task) + task["checks"][0]["required_to_pass"] = True + evaluation = evaluate_branch_review(task, self.workspace, self.review, [required]) + self.assertFalse(evaluation["accept_allowed"]) + self.assertEqual(evaluation["required_failures"], ["fixture-tests"]) + with self.assertRaisesRegex(ContractError, "blocked by required evidence"): + apply_lead_disposition( + evaluation, "accept", revisions_used=0, max_revisions=2, + ) + + def test_revise_is_bounded_and_reject_is_terminal(self): + evaluation = evaluate_branch_review( + self.task, self.workspace, self.review, [self.check], + ) + revised = apply_lead_disposition( + evaluation, "revise", revisions_used=0, max_revisions=1, + ) + self.assertEqual( + (revised["action"], revised["terminal_state"], revised["next_revision"]), + ("repeat_review", None, 1), + ) + exhausted = apply_lead_disposition( + evaluation, "revise", revisions_used=1, max_revisions=1, + ) + self.assertEqual( + (exhausted["action"], exhausted["terminal_state"]), + ("budget_exhausted", "failed"), + ) + rejected = apply_lead_disposition( + evaluation, "reject", revisions_used=0, max_revisions=1, + ) + self.assertEqual(rejected["terminal_state"], "failed") + + def test_prompt_is_deterministic_read_only_and_distinguishes_adversarial_mode(self): + first = build_review_prompt(self.task, self.workspace) + second = build_review_prompt(copy.deepcopy(self.task), copy.deepcopy(self.workspace)) + self.assertEqual(first, second) + self.assertIn("read-only reviewer", first) + self.assertIn('"candidate_sha256":"' + "c" * 64 + '"', first) + adversarial = copy.deepcopy(self.task) + adversarial["review"] = {"mode": "adversarial", "focus": "trust boundaries"} + prompt = build_review_prompt(adversarial, self.workspace) + self.assertIn('"review_mode":"adversarial"', prompt) + self.assertIn('"focus":"trust boundaries"', prompt) + self.assertNotEqual(prompt, first) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_validation.py b/test/core/test_validation.py index 0b60328..5492ecd 100644 --- a/test/core/test_validation.py +++ b/test/core/test_validation.py @@ -49,6 +49,8 @@ def test_task_rejects_adversarial_nested_types(self): lambda t: t["budget"].__setitem__("wall_seconds", True), lambda t: t["scope"]["read_paths"].append("../escape"), lambda t: t["origin"].__setitem__("session_ref", []), + lambda t: t["acceptance"].append(copy.deepcopy(t["acceptance"][0])), + lambda t: t["checks"].append(copy.deepcopy(t["checks"][0])), ] for mutate in mutations: value = copy.deepcopy(self.task); mutate(value) From 97d2c6cc197c63a4fe10360e2176eb746dd1ef80 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 16 Sep 2026 01:34:41 +0530 Subject: [PATCH 052/197] feat: run durable M3 review evidence pipeline --- plugin/core/schemas/check-result.schema.json | 2 +- plugin/core/schemas/task.schema.json | 6 +- plugin/core/src/devsquad/detached.py | 50 +++- plugin/core/src/devsquad/review_worker.py | 180 ++++++++++++ plugin/core/src/devsquad/service.py | 50 +++- plugin/core/src/devsquad/store.py | 285 +++++++++++++++---- plugin/core/src/devsquad/supervisor.py | 78 ++++- plugin/core/src/devsquad/validation.py | 12 +- plugin/core/src/devsquad/workflows.py | 141 ++++++++- plugin/core/src/devsquad/workspaces.py | 61 +++- test/core/test_handoff_store.py | 76 +++++ test/core/test_review_runtime.py | 188 ++++++++++++ test/core/test_review_workflow.py | 42 +++ test/core/test_service.py | 4 + test/core/test_validation.py | 9 + test/core/test_workspaces.py | 22 ++ 16 files changed, 1119 insertions(+), 87 deletions(-) create mode 100644 plugin/core/src/devsquad/review_worker.py create mode 100644 test/core/test_review_runtime.py diff --git a/plugin/core/schemas/check-result.schema.json b/plugin/core/schemas/check-result.schema.json index 0e782e1..ab09628 100644 --- a/plugin/core/schemas/check-result.schema.json +++ b/plugin/core/schemas/check-result.schema.json @@ -25,7 +25,7 @@ "additionalProperties": false, "required": ["preview", "captured_bytes", "total_bytes", "truncated", "full_sha256"], "properties": { - "preview": {"type": "string", "maxLength": 65536}, + "preview": {"type": "string", "maxLength": 1024}, "captured_bytes": {"type": "integer", "minimum": 0}, "total_bytes": {"type": "integer", "minimum": 0}, "truncated": {"type": "boolean"}, diff --git a/plugin/core/schemas/task.schema.json b/plugin/core/schemas/task.schema.json index 96c9830..9b129d5 100644 --- a/plugin/core/schemas/task.schema.json +++ b/plugin/core/schemas/task.schema.json @@ -5,8 +5,8 @@ "properties": { "schema_version": {"const": 1}, "workflow": {"enum": ["branch-review", "issue-delivery"]}, "goal": {"type": "string", "minLength": 1}, "task_class": {"type": "string", "minLength": 1}, - "project": {"$ref": "#/$defs/project"}, "acceptance": {"type": "array", "minItems": 1, "items": {"$ref": "#/$defs/acceptance"}}, - "checks": {"type": "array", "items": {"$ref": "#/$defs/check"}}, "scope": {"$ref": "#/$defs/scope"}, + "project": {"$ref": "#/$defs/project"}, "acceptance": {"type": "array", "minItems": 1, "maxItems": 100, "items": {"$ref": "#/$defs/acceptance"}}, + "checks": {"type": "array", "maxItems": 16, "items": {"$ref": "#/$defs/check"}}, "scope": {"$ref": "#/$defs/scope"}, "lead": {"$ref": "#/$defs/lead"}, "routing": {"$ref": "#/$defs/routing"}, "budget": {"$ref": "#/$defs/budget"}, "origin": {"$ref": "#/$defs/origin"}, "review": {"$ref": "#/$defs/review"} }, @@ -14,7 +14,7 @@ "project": {"type":"object","additionalProperties":false,"required":["repo_path","base_ref","target_ref"],"properties":{"repo_path":{"type":"string","pattern":"^/"},"base_ref":{"type":"string","minLength":1},"target_ref":{"type":"string","minLength":1}}}, "acceptance": {"type":"object","additionalProperties":false,"required":["id","description","evidence_kind"],"properties":{"id":{"type":"string","minLength":1},"description":{"type":"string","minLength":1},"evidence_kind":{"enum":["review","check","artifact","host"]}}}, "check": {"type":"object","additionalProperties":false,"required":["id","argv","cwd","timeout_seconds","required_to_pass"],"properties":{"id":{"type":"string","minLength":1},"argv":{"type":"array","minItems":1,"items":{"type":"string","minLength":1}},"cwd":{"type":"string"},"timeout_seconds":{"type":"integer","minimum":1},"required_to_pass":{"type":"boolean"}}}, - "scope": {"type":"object","additionalProperties":false,"required":["read_paths","write_paths"],"properties":{"read_paths":{"type":"array","uniqueItems":true,"items":{"type":"string","minLength":1}},"write_paths":{"type":"array","uniqueItems":true,"items":{"type":"string","minLength":1}}}}, + "scope": {"type":"object","additionalProperties":false,"required":["read_paths","write_paths"],"properties":{"read_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}},"write_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}}}}, "lead": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["host","headless"]}}}, "override": {"type":"object","additionalProperties":false,"required":["profile_id"],"properties":{"profile_id":{"type":"string","minLength":1},"fallback":{"enum":["none","policy"]}}}, "routing": {"type":"object","additionalProperties":false,"required":["profiles_file","policy_file"],"properties":{"profiles_file":{"type":"string","minLength":1},"policy_file":{"type":"string","minLength":1},"overrides":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"$ref":"#/$defs/override"}}}}, diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index 1f89c0b..ceece57 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -6,7 +6,7 @@ import sys from .contracts import ExecutionIdentity, LaunchSpec -from .store import ConflictError, Store +from .store import ConflictError, Store, canonical_json from .supervisor import Supervisor @@ -22,11 +22,49 @@ def main(argv=None): if run.get("package_digest") != args.package_digest: raise ConflictError("detached package digest does not match prepared run") snapshot = json.loads(run["mutable_snapshot"]) - environment = {"DEVSQUAD_WORKER": "1", "DEVSQUAD_RUN_ID": args.run_id} - identity = ExecutionIdentity("devsquad-fake-step", "1", None, None, None, None) - command = [sys.executable, "-P", "-m", "devsquad.fake_step"] - if "internal_fake_delay" in snapshot: command += ["--delay", str(snapshot["internal_fake_delay"])] - spec = LaunchSpec(1, "devsquad-fake-step", "cli_exec", tuple(command), run["worktree_path"], None, snapshot["task"]["budget"]["wall_seconds"], identity, environment) + environment = { + "DEVSQUAD_WORKER": "1", + "DEVSQUAD_RUN_ID": args.run_id, + "DEVSQUAD_DELEGATION_DEPTH": "1", + } + stdin_path = None + if "internal_review_fixture" in snapshot: + selected = snapshot["routing"]["roles"]["reviewer"]["selected"]["profile"] + identity = ExecutionIdentity( + "devsquad-review-workflow", + "1", + None, + selected["model_family"], + selected["model_id"], + selected["effort"]["value"], + tuple(selected["required_tools"]), + selected["permission_policy"], + selected["account_pool_id"], + "unknown", + ) + command = [sys.executable, "-P", "-m", "devsquad.review_worker"] + input_path, _, _ = store.finalize_artifact( + args.run_id, + "workflow-input.json", + canonical_json(snapshot).encode(), + ) + stdin_path = str(input_path) + else: + identity = ExecutionIdentity("devsquad-fake-step", "1", None, None, None, None) + command = [sys.executable, "-P", "-m", "devsquad.fake_step"] + if "internal_fake_delay" in snapshot: + command += ["--delay", str(snapshot["internal_fake_delay"])] + spec = LaunchSpec( + 1, + identity.harness, + "cli_exec", + tuple(command), + run["worktree_path"], + stdin_path, + snapshot["task"]["budget"]["wall_seconds"], + identity, + environment, + ) supervisor = Supervisor(store) try: handle = supervisor.launch_durable(args.run_id, args.expected_version, spec, f"daemon:{os.getpid()}", args.package_digest) except ConflictError: return 0 diff --git a/plugin/core/src/devsquad/review_worker.py b/plugin/core/src/devsquad/review_worker.py new file mode 100644 index 0000000..217d392 --- /dev/null +++ b/plugin/core/src/devsquad/review_worker.py @@ -0,0 +1,180 @@ +"""Run one frozen branch-review fixture and its trusted declared checks.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import time +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .supervisor import BoundedDrain +from .workflows import MAX_PREVIEW_CHARS, make_branch_review_evidence +from .workspaces import dirty_paths + + +MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 + + +def _empty_stream() -> dict[str, Any]: + return { + "preview": "", + "captured_bytes": 0, + "total_bytes": 0, + "truncated": False, + "full_sha256": hashlib.sha256(b"").hexdigest(), + } + + +def _stream_result(drain: BoundedDrain) -> dict[str, Any]: + metadata = drain.finish() + return { + "preview": bytes(drain.content).decode("utf-8", "replace"), + "captured_bytes": metadata["captured_bytes"], + "total_bytes": metadata["total_bytes"], + "truncated": metadata["truncated"], + "full_sha256": metadata["full_sha256"], + } + + +def _safe_check_cwd(root: Path, relative: str) -> Path: + try: + candidate = (root / relative).resolve(strict=True) + except OSError as exc: + raise ContractError(f"declared check cwd does not exist: {relative}") from exc + if (candidate != root and root not in candidate.parents) or not candidate.is_dir(): + raise ContractError(f"declared check cwd escapes the check workspace: {relative}") + return candidate + + +def _run_check( + check: dict[str, Any], + check_workspace: Path, + candidate_sha256: str, + target_oid: str, +) -> dict[str, Any]: + started = time.monotonic() + stdout_result, stderr_result = _empty_stream(), _empty_stream() + cwd = _safe_check_cwd(check_workspace, check["cwd"]) + try: + process = subprocess.Popen( + check["argv"], + cwd=cwd, + env=os.environ.copy(), + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + start_new_session=False, + close_fds=True, + ) + except OSError: + status, returncode, error_code = "launch_failed", None, "CLI_ERROR" + else: + assert process.stdout is not None and process.stderr is not None + stdout = BoundedDrain(process.stdout, MAX_PREVIEW_CHARS) + stderr = BoundedDrain(process.stderr, MAX_PREVIEW_CHARS) + stdout.start() + stderr.start() + try: + returncode = process.wait(timeout=check["timeout_seconds"]) + status = "passed" if returncode == 0 else "failed" + error_code = None + except subprocess.TimeoutExpired: + status, returncode, error_code = "timed_out", None, "TIMEOUT" + process.terminate() + try: + process.wait(timeout=1) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=2) + stdout_result, stderr_result = _stream_result(stdout), _stream_result(stderr) + return { + "schema_version": 1, + "candidate_sha256": candidate_sha256, + "target_oid": target_oid, + "id": check["id"], + "argv": check["argv"], + "cwd": check["cwd"], + "required_to_pass": check["required_to_pass"], + "status": status, + "returncode": returncode, + "error_code": error_code, + "duration_ms": max(0, int((time.monotonic() - started) * 1000)), + "stdout": stdout_result, + "stderr": stderr_result, + } + + +def _verify_finding_locations(review: dict[str, Any], workspace: Path) -> None: + for finding in review["findings"]: + try: + target = (workspace / finding["path"]).resolve(strict=True) + except OSError as exc: + raise ContractError( + f"review finding path does not exist: {finding['path']}" + ) from exc + if target != workspace and workspace not in target.parents: + raise ContractError("review finding path escapes the frozen workspace") + if not target.is_file(): + raise ContractError(f"review finding path is not a file: {finding['path']}") + lines = 0 + with target.open("rb") as stream: + for lines, _ in enumerate(stream, 1): + if lines >= finding["end_line"]: + break + if lines < finding["end_line"]: + raise ContractError( + f"review finding line is outside the frozen file: {finding['path']}" + ) + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("workflow snapshot must be an object") + task = snapshot.get("task") + workspace = snapshot.get("workspace") + check_workspace = snapshot.get("check_workspace") + fixture = snapshot.get("internal_review_fixture") + if not all(isinstance(value, dict) for value in ( + task, workspace, check_workspace, fixture, + )): + raise ContractError("offline review snapshot is incomplete") + review_root = Path(workspace["path"]).resolve(strict=True) + checks_root = Path(check_workspace["path"]).resolve(strict=True) + if dirty_paths(review_root): + raise ContractError("frozen review workspace is dirty before reviewer execution") + review = fixture + _verify_finding_locations(review, review_root) + if dirty_paths(review_root): + raise ContractError("reviewer modified the frozen read-only workspace") + checks = [ + _run_check( + check, + checks_root, + workspace["candidate_sha256"], + workspace["target_oid"], + ) + for check in task["checks"] + ] + return make_branch_review_evidence(snapshot, review, checks) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_SNAPSHOT_BYTES + 1) + if len(payload) > MAX_SNAPSHOT_BYTES: + raise ContractError("workflow snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("workflow snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 54cf7ea..043b72c 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -24,9 +24,11 @@ canonical_json, ) from .validation import validate_task +from .workflows import review_mode, validate_review_document from .workspaces import ( assert_clean_inputs, committed_regular_file, + prepare_check_workspace, prepare_review_workspace, repo_relative_config, resolve_commit, @@ -162,6 +164,7 @@ def _resolve_snapshot( *, project_id: str | None = None, run_id: str | None = None, + internal_review_fixture: dict[str, Any] | None = None, ) -> dict[str, Any]: repo = resolved_repo or Path(task["project"]["repo_path"]).resolve(strict=True) base_oid = resolve_commit(repo, task["project"]["base_ref"]) @@ -214,6 +217,31 @@ def _resolve_snapshot( scope_paths, required_clean_paths=config_paths.values(), ) + snapshot["check_workspace"] = prepare_check_workspace( + repo, + self.runtime, + project_id, + run_id, + target_oid, + scope_paths, + required_clean_paths=config_paths.values(), + ) + if internal_review_fixture is not None: + if (not isinstance(internal_review_fixture, dict) + or set(internal_review_fixture) + != {"verdict", "summary", "findings"}): + raise ContractError("internal review fixture fields are invalid") + fixture_document = { + "schema_version": 1, + "candidate_sha256": snapshot["workspace"]["candidate_sha256"], + "base_oid": base_oid, + "target_oid": target_oid, + "review_mode": review_mode(task), + **internal_review_fixture, + } + snapshot["internal_review_fixture"] = validate_review_document( + fixture_document, task, snapshot["workspace"], + ) return snapshot def _continue_preparation( @@ -229,6 +257,7 @@ def _continue_preparation( try: task = submitted["task"] internal_delay = submitted.get("_internal_fake_delay") + internal_review_fixture = submitted.get("_internal_review_fixture") store.validate_predecessor(run_id, fencing_token, supersedes_run_id) validated_supersedes_run_id = supersedes_run_id validate_task(task, require_existing_repo=True) @@ -242,8 +271,9 @@ def _continue_preparation( worktree, project_id=project_id, run_id=run_id, + internal_review_fixture=internal_review_fixture, ) - if internal_delay is None: + if internal_delay is None and internal_review_fixture is None: error = { "error": "CAPABILITY_UNAVAILABLE", "message": "branch-review workflow is introduced in M3", @@ -264,6 +294,10 @@ def _continue_preparation( package_path=str(package), package_digest=digest, supersedes_run_id=supersedes_run_id, + worktree_path=( + snapshot["workspace"]["path"] + if internal_review_fixture is not None else None + ), ) return (version, package, digest), None except Exception as exc: @@ -283,11 +317,23 @@ def _continue_preparation( raise exc return None, error - def start(self, task: dict[str, Any], idempotency_key: str, supersedes_run_id: str | None = None, *, _internal_fake_delay: float | None = None) -> dict[str, Any]: + def start( + self, + task: dict[str, Any], + idempotency_key: str, + supersedes_run_id: str | None = None, + *, + _internal_fake_delay: float | None = None, + _internal_review_fixture: dict[str, Any] | None = None, + ) -> dict[str, Any]: validate_task(task, require_existing_repo=True) + if _internal_fake_delay is not None and _internal_review_fixture is not None: + raise ContractError("internal lifecycle fixtures are mutually exclusive") submitted = {"task": task, "supersedes_run_id": supersedes_run_id} if _internal_fake_delay is not None: submitted["_internal_fake_delay"] = _internal_fake_delay + if _internal_review_fixture is not None: + submitted["_internal_review_fixture"] = _internal_review_fixture store = self._store() try: claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}") diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index b09f092..9fd8779 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -708,20 +708,19 @@ def record_attempt_output(self, run_id: str, attempt_token: str, stdout_id: str, self.connection.execute("ROLLBACK") raise - def commit_durable_import(self, run_id: str, attempt_token: str, artifacts: list[dict[str, Any]], metadata: Any, terminal_state: str, payload: Any) -> str: - """Atomically import one durable receipt, or observe its prior import. - - Content-addressed files are finalized before this call. All database - references, output projection fields, and terminal state then cross a - single write fence so competing recovery processes cannot partially - import or downgrade a valid completion. - """ - if terminal_state not in TERMINAL_STATES: - raise ContractError("invalid terminal state") + def _prepare_durable_artifacts( + self, + run_id: str, + artifacts: list[dict[str, Any]], + *, + require_result_receipt: bool, + ) -> tuple[list[tuple[str, str, str, int]], str, str]: expected_parent = (self.artifacts / run_id).resolve() prepared = [] names = set() for artifact in artifacts: + if not isinstance(artifact, dict): + raise ContractError("durable artifacts must be objects") name = artifact.get("name") path = Path(artifact.get("path", "")) digest = artifact.get("sha256") @@ -735,10 +734,104 @@ def commit_durable_import(self, run_id: str, attempt_token: str, artifacts: list if hashlib.sha256(content).hexdigest() != digest or len(content) != size: raise ConflictError("durable artifact changed before database import") prepared.append((name, str(path), digest, size)) - stdout_name = next((name for name in names if name.endswith(".stdout")), None) - stderr_name = next((name for name in names if name.endswith(".stderr")), None) - if stdout_name is None or stderr_name is None or "result-receipt.json" not in names: - raise ContractError("durable import requires stdout, stderr, and result receipt artifacts") + stdout_names = [name for name in names if name.endswith(".stdout")] + stderr_names = [name for name in names if name.endswith(".stderr")] + if (len(stdout_names) != 1 or len(stderr_names) != 1 + or (require_result_receipt and "result-receipt.json" not in names)): + requirement = "stdout, stderr, and result receipt" if require_result_receipt else "stdout and stderr" + raise ContractError(f"durable import requires exactly one {requirement}") + return prepared, stdout_names[0], stderr_names[0] + + def _reference_prepared_artifacts( + self, + run_id: str, + version: int, + prepared: list[tuple[str, str, str, int]], + ) -> tuple[int, dict[str, str]]: + artifact_ids = {} + for name, path, digest, size in prepared: + existing = self.connection.execute( + "SELECT id,path,sha256,byte_size FROM artifacts WHERE run_id=? AND name=?", + (run_id, name), + ).fetchone() + if existing: + if (existing["path"] != path or existing["sha256"] != digest + or existing["byte_size"] != size): + raise ConflictError("durable artifact conflicts with an existing reference") + artifact_ids[name] = existing["id"] + continue + artifact_id, now = str(uuid.uuid4()), _utc_now() + self.connection.execute( + "INSERT INTO artifacts(id,run_id,name,path,sha256,byte_size,created_at) " + "VALUES(?,?,?,?,?,?,?)", + (artifact_id, run_id, name, path, digest, size, now), + ) + artifact_ids[name] = artifact_id + version += 1 + event = canonical_json({ + "artifact_id": artifact_id, + "name": name, + "sha256": digest, + "byte_size": size, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'artifact.recorded',?,?)", + (run_id, version, event, now), + ) + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + return version, artifact_ids + + def _record_prepared_output( + self, + run_id: str, + version: int, + attempt: sqlite3.Row, + artifact_ids: dict[str, str], + stdout_name: str, + stderr_name: str, + encoded_metadata: str, + ) -> int: + stdout_id, stderr_id = artifact_ids[stdout_name], artifact_ids[stderr_name] + if attempt["output_metadata"] is None: + now = _utc_now() + version += 1 + self.connection.execute( + "UPDATE attempts SET stdout_artifact_id=?,stderr_artifact_id=?," + "output_metadata=? WHERE id=?", + (stdout_id, stderr_id, encoded_metadata, attempt["id"]), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'attempt.output',?,?)", + (run_id, version, encoded_metadata, now), + ) + self.connection.execute( + "UPDATE runs SET version=?,updated_at=? WHERE id=?", + (version, now, run_id), + ) + elif (attempt["stdout_artifact_id"] != stdout_id + or attempt["stderr_artifact_id"] != stderr_id + or attempt["output_metadata"] != encoded_metadata): + raise ConflictError("durable output conflicts with its prior import") + return version + + def commit_durable_import(self, run_id: str, attempt_token: str, artifacts: list[dict[str, Any]], metadata: Any, terminal_state: str, payload: Any) -> str: + """Atomically import one durable receipt, or observe its prior import. + + Content-addressed files are finalized before this call. All database + references, output projection fields, and terminal state then cross a + single write fence so competing recovery processes cannot partially + import or downgrade a valid completion. + """ + if terminal_state not in TERMINAL_STATES: + raise ContractError("invalid terminal state") + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=True, + ) encoded_metadata, encoded_payload = canonical_json(metadata), canonical_json(payload) self.connection.execute("BEGIN IMMEDIATE") @@ -756,51 +849,18 @@ def commit_durable_import(self, run_id: str, attempt_token: str, artifacts: list if attempt["status"] not in {"running", "cancelling"} or run["state"] not in {"running", "cancelling"}: raise ConflictError("durable import is fenced") - version = run["version"] - artifact_ids = {} - for name, path, digest, size in prepared: - existing = self.connection.execute( - "SELECT id,path,sha256,byte_size FROM artifacts WHERE run_id=? AND name=?", (run_id, name), - ).fetchone() - if existing: - if existing["path"] != path or existing["sha256"] != digest or existing["byte_size"] != size: - raise ConflictError("durable artifact conflicts with an existing reference") - artifact_ids[name] = existing["id"] - continue - artifact_id, now = str(uuid.uuid4()), _utc_now() - self.connection.execute( - "INSERT INTO artifacts(id,run_id,name,path,sha256,byte_size,created_at) VALUES(?,?,?,?,?,?,?)", - (artifact_id, run_id, name, path, digest, size, now), - ) - artifact_ids[name] = artifact_id - version += 1 - event = canonical_json({"artifact_id": artifact_id, "name": name, "sha256": digest, "byte_size": size}) - self.connection.execute( - "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'artifact.recorded',?,?)", - (run_id, version, event, now), - ) - self.connection.execute( - "UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id), - ) - - stdout_id, stderr_id = artifact_ids[stdout_name], artifact_ids[stderr_name] - if attempt["output_metadata"] is None: - now = _utc_now(); version += 1 - self.connection.execute( - "UPDATE attempts SET stdout_artifact_id=?,stderr_artifact_id=?,output_metadata=? WHERE id=?", - (stdout_id, stderr_id, encoded_metadata, attempt["id"]), - ) - self.connection.execute( - "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'attempt.output',?,?)", - (run_id, version, encoded_metadata, now), - ) - self.connection.execute( - "UPDATE runs SET version=?,updated_at=? WHERE id=?", (version, now, run_id), - ) - elif (attempt["stdout_artifact_id"] != stdout_id - or attempt["stderr_artifact_id"] != stderr_id - or attempt["output_metadata"] != encoded_metadata): - raise ConflictError("durable output conflicts with its prior import") + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) now = _utc_now(); version += 1 effective_state = "cancelled" if run["state"] == "cancelling" else terminal_state @@ -820,6 +880,113 @@ def commit_durable_import(self, run_id: str, attempt_token: str, artifacts: list self.connection.execute("ROLLBACK") raise + def commit_durable_handoff( + self, + run_id: str, + attempt_token: str, + artifacts: list[dict[str, Any]], + metadata: Any, + packet: dict[str, Any], + ) -> str: + """Atomically import one completed attempt and publish its host packet.""" + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=False, + ) + encoded_metadata = canonical_json(metadata) + frozen_packet = json.loads(canonical_json(packet)) + packet_artifacts = frozen_packet.get("artifacts") + if not isinstance(packet_artifacts, list) or not packet_artifacts: + raise ContractError("handoff packet must reference evidence artifacts") + prepared_hashes = {name: digest for name, _, digest, _ in prepared} + referenced_names = set() + for reference in packet_artifacts: + if not isinstance(reference, dict) or set(reference) != {"name", "sha256"}: + raise ContractError("handoff artifact references are invalid") + name, digest = reference["name"], reference["sha256"] + if (not isinstance(name, str) or name in referenced_names + or prepared_hashes.get(name) != digest): + raise ContractError("handoff artifact does not match durable evidence") + referenced_names.add(name) + + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,stdout_artifact_id,stderr_artifact_id,output_metadata " + "FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("durable handoff import is fenced") + if attempt["status"] == "finished" and run["state"] == "awaiting_host": + self.connection.execute("COMMIT") + return "awaiting_host" + if (attempt["status"] != "running" or run["state"] != "running" + or run["phase"] is not None): + raise ConflictError("durable handoff import is fenced") + + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) + for reference in packet_artifacts: + reference["artifact_id"] = artifact_ids[reference["name"]] + packet_json = canonical_json(frozen_packet) + packet_sha256 = hashlib.sha256(packet_json.encode()).hexdigest() + if self.connection.execute( + "SELECT 1 FROM handoffs WHERE run_id=? AND status IN ('open','submitted')", + (run_id,), + ).fetchone(): + raise ConflictError("run already has a pending handoff") + sequence = self.connection.execute( + "SELECT COALESCE(MAX(sequence),0)+1 FROM handoffs WHERE run_id=?", + (run_id,), + ).fetchone()[0] + handoff_id, now = str(uuid.uuid4()), _utc_now() + version += 1 + self.connection.execute( + "INSERT INTO handoffs(id,run_id,sequence,packet_json,packet_sha256,status," + "created_run_version,created_at) VALUES(?,?,?,?,?,'open',?,?)", + (handoff_id, run_id, sequence, packet_json, packet_sha256, version, now), + ) + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='awaiting_host',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff_id, + "packet_sha256": packet_sha256, + "sequence": sequence, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.awaiting_host',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return "awaiting_host" + except Exception: + self.connection.execute("ROLLBACK") + raise + def block_recovery(self, run_id: str, attempt_token: str, reason: str, *, release_writer: bool = False) -> int: self.connection.execute("BEGIN IMMEDIATE") try: diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 23541a2..3a3efb8 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -18,6 +18,7 @@ from .contracts import ContractError, LaunchSpec from .store import AttemptReservation, ConflictError, Store, canonical_json +from .workflows import decode_branch_review_evidence def _open_stdin_artifact(path: str) -> BinaryIO: @@ -284,6 +285,63 @@ def wait_durable(self, handle: DurableAttempt, timeout_seconds: float) -> int: self.import_durable(handle.reservation.run_id) return returncode + def _commit_review_handoff( + self, + run_id: str, + attempt: dict[str, Any], + stream_artifacts: list[dict[str, Any]], + metadata: dict[str, Any], + snapshot: dict[str, Any], + stdout: bytes, + ) -> str: + evidence = decode_branch_review_evidence(stdout, snapshot) + documents = { + "review.json": evidence["review"], + "checks.json": { + "schema_version": 1, + "candidate_sha256": evidence["candidate_sha256"], + "target_oid": evidence["target_oid"], + "results": evidence["checks"], + }, + "evaluation.json": evidence["evaluation"], + "review-attempt.json": evidence["attempt"], + } + artifacts = list(stream_artifacts) + evidence_references = [] + for name, document in documents.items(): + content = (canonical_json(document) + "\n").encode() + path, digest, size = self.store.finalize_artifact(run_id, name, content) + artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + evidence_references.append({"name": name, "sha256": digest}) + packet = { + "schema_version": 1, + "workflow": "branch-review", + "candidate_sha256": evidence["candidate_sha256"], + "base_oid": evidence["base_oid"], + "target_oid": evidence["target_oid"], + "review": evidence["review"], + "checks": evidence["checks"], + "evaluation": evidence["evaluation"], + "attempt": evidence["attempt"], + "artifacts": evidence_references, + "instructions": ( + "Inspect the bound review and check evidence, then submit exactly one " + "accept, revise, or reject disposition." + ), + } + return self.store.commit_durable_handoff( + run_id, + attempt["attempt_token"], + artifacts, + metadata, + packet, + ) + def import_durable(self, run_id: str) -> str: attempt=self.store.attempt(run_id) if not attempt or attempt["status"] not in {"running","cancelling"}: @@ -333,19 +391,37 @@ def import_durable(self, run_id: str) -> str: raise ValueError("invalid receipt") metadata={name:receipt[name] for name in ("stdout","stderr")} artifacts=[] + captures={} for stream,column in (("stdout","stdout_spool"),("stderr","stderr_spool")): data=Path(attempt[column]).read_bytes(); meta=metadata[stream] if hashlib.sha256(data).hexdigest()!=meta["captured_sha256"] or len(data)!=meta["captured_bytes"]: raise ValueError("capture hash mismatch") + captures[stream]=data logical=f"{attempt['id']}.{stream}" path,digest,size=self.store.finalize_artifact(run_id,logical,data) artifacts.append({"name":logical,"path":path,"sha256":digest,"byte_size":size}) + snapshot=json.loads(self.store.run(run_id)["mutable_snapshot"]) + workflow_review="internal_review_fixture" in snapshot + semantic_error=None + if (workflow_review and not receipt["cancelled"] + and not receipt["timed_out"] and receipt["returncode"]==0): + try: + return self._commit_review_handoff( + run_id, attempt, artifacts, metadata, snapshot, captures["stdout"], + ) + except ContractError as exc: + semantic_error=str(exc) + receipt["error"]="WORKFLOW_OUTPUT_INVALID" + receipt["message"]=semantic_error receipt_bytes=canonical_json(receipt).encode() path,digest,size=self.store.finalize_artifact(run_id,"result-receipt.json",receipt_bytes) artifacts.append({"name":"result-receipt.json","path":path,"sha256":digest,"byte_size":size}) - terminal="cancelled" if receipt["cancelled"] else ("failed" if receipt["timed_out"] or receipt["returncode"]!=0 else "succeeded") + terminal="cancelled" if receipt["cancelled"] else ("failed" if receipt["timed_out"] or receipt["returncode"]!=0 or semantic_error else "succeeded") payload={"returncode":receipt["returncode"],"receipt":"result-receipt.json"} if receipt["timed_out"]: payload["error"]="TIMEOUT" + if semantic_error: + payload["error"]="WORKFLOW_OUTPUT_INVALID" + payload["message"]=semantic_error return self.store.commit_durable_import( run_id,attempt["attempt_token"],artifacts,metadata,terminal,payload, ) diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py index ee51702..2647879 100644 --- a/plugin/core/src/devsquad/validation.py +++ b/plugin/core/src/devsquad/validation.py @@ -6,6 +6,9 @@ from .contracts import ContractError TASK_FIELDS = {"schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin", "review"} +MAX_ACCEPTANCE_CRITERIA = 100 +MAX_CHECKS = 16 +MAX_SCOPE_PATHS = 256 def _exact(value: dict[str, Any], allowed: set[str], required: set[str], label: str) -> None: @@ -34,8 +37,9 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) raise ContractError("project.repo_path must be an existing absolute Git repository") if not isinstance(value["goal"], str) or not value["goal"].strip() or not isinstance(value["task_class"], str) or not value["task_class"].strip(): raise ContractError("goal and task_class must be non-empty strings") - if not isinstance(value["acceptance"], list) or not value["acceptance"]: - raise ContractError("acceptance must be non-empty") + if (not isinstance(value["acceptance"], list) or not value["acceptance"] + or len(value["acceptance"]) > MAX_ACCEPTANCE_CRITERIA): + raise ContractError("acceptance must be a non-empty bounded array") acceptance_ids = set() for item in value["acceptance"]: _exact(item, {"id", "description", "evidence_kind"}, {"id", "description", "evidence_kind"}, "acceptance item") @@ -46,7 +50,8 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) if item["id"] in acceptance_ids: raise ContractError("acceptance ids must be unique") acceptance_ids.add(item["id"]) - if not isinstance(value["checks"], list): raise ContractError("checks must be an array") + if not isinstance(value["checks"], list) or len(value["checks"]) > MAX_CHECKS: + raise ContractError("checks must be a bounded array") check_ids = set() for check in value["checks"]: _exact(check, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, "check") @@ -59,6 +64,7 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) if type(check["required_to_pass"]) is not bool: raise ContractError("required_to_pass must be boolean") scope = value["scope"]; _exact(scope, {"read_paths", "write_paths"}, {"read_paths", "write_paths"}, "scope") if not isinstance(scope["read_paths"], list) or not isinstance(scope["write_paths"], list) or not all(isinstance(p, str) and p for p in scope["read_paths"] + scope["write_paths"]): raise ContractError("scope paths must be non-empty string arrays") + if len(scope["read_paths"]) + len(scope["write_paths"]) > MAX_SCOPE_PATHS: raise ContractError("scope paths exceed their bound") if len(set(scope["read_paths"])) != len(scope["read_paths"]) or len(set(scope["write_paths"])) != len(scope["write_paths"]): raise ContractError("scope paths must be unique") for p in scope["read_paths"] + scope["write_paths"]: _relative(p, "scope path") if value["workflow"] == "branch-review" and scope["write_paths"]: raise ContractError("branch review cannot write") diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index 3e2b856..cdd4def 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -2,6 +2,7 @@ from __future__ import annotations +import hashlib import json from pathlib import PurePosixPath import re @@ -13,9 +14,10 @@ MAX_REVIEW_BYTES = 512 * 1024 +MAX_EVIDENCE_BYTES = 1024 * 1024 MAX_FINDINGS = 100 MAX_TEXT_CHARS = 20_000 -MAX_PREVIEW_CHARS = 64 * 1024 +MAX_PREVIEW_CHARS = 1024 FINDING_SEVERITIES = {"critical", "high", "medium", "low"} CHECK_STATUSES = {"passed", "failed", "timed_out", "launch_failed"} REVIEW_MODES = {"standard", "adversarial"} @@ -79,14 +81,19 @@ def _inside_scope(path: str, scopes: list[str]) -> bool: return False -def _strict_json_object(payload: bytes | str, label: str) -> dict[str, Any]: +def _strict_json_object( + payload: bytes | str, + label: str, + *, + maximum: int = MAX_REVIEW_BYTES, +) -> dict[str, Any]: if isinstance(payload, str): encoded = payload.encode() elif isinstance(payload, bytes): encoded = payload else: raise ContractError(f"{label} must be UTF-8 JSON bytes or text") - if len(encoded) > MAX_REVIEW_BYTES: + if len(encoded) > maximum: raise ContractError(f"{label} exceeds its byte limit") def object_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]: @@ -392,3 +399,131 @@ def build_review_prompt(task: dict[str, Any], workspace: dict[str, Any]) -> str: "Frozen assignment:", canonical_json(assignment), ]) + + +def validate_branch_review_evidence( + value: dict[str, Any], + snapshot: dict[str, Any], +) -> dict[str, Any]: + """Recompute all derived gates before a worker result can become evidence.""" + document = _exact(value, { + "schema_version", "workflow", "candidate_sha256", "base_oid", "target_oid", + "review", "checks", "evaluation", "attempt", + }, "branch review evidence") + if document["schema_version"] != 1 or type(document["schema_version"]) is not int: + raise ContractError("branch review evidence schema_version is invalid") + if document["workflow"] != "branch-review": + raise ContractError("branch review evidence workflow is invalid") + if not isinstance(snapshot, dict): + raise ContractError("frozen workflow snapshot must be an object") + task, workspace = snapshot.get("task"), snapshot.get("workspace") + if not isinstance(task, dict) or not isinstance(workspace, dict): + raise ContractError("frozen workflow snapshot is incomplete") + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + for field, expected, validator in ( + ("candidate_sha256", candidate_sha256, _sha256), + ("base_oid", base_oid, _commit_oid), + ("target_oid", target_oid, _commit_oid), + ): + if validator(document[field], f"evidence {field}") != expected: + raise ContractError(f"branch review evidence changes frozen {field}") + review = validate_review_document(document["review"], task, workspace) + checks = validate_check_results(document["checks"], task, workspace) + expected_evaluation = evaluate_branch_review(task, workspace, review, checks) + if canonical_json(document["evaluation"]) != canonical_json(expected_evaluation): + raise ContractError("branch review evaluation does not match derived gates") + attempt = _exact(document["attempt"], { + "role", "selected_profile", "prompt_sha256", "review_sha256", + "worker_invocations", "native_model_requests", "usage", + }, "review attempt evidence") + if attempt["role"] != "reviewer": + raise ContractError("review attempt role is invalid") + try: + frozen_reviewer = snapshot["routing"]["roles"]["reviewer"]["selected"] + except (KeyError, TypeError) as exc: + raise ContractError("frozen reviewer selection is missing") from exc + if canonical_json(attempt["selected_profile"]) != canonical_json(frozen_reviewer): + raise ContractError("review attempt changes the frozen selected profile") + prompt_sha256 = hashlib.sha256(build_review_prompt(task, workspace).encode()).hexdigest() + if _sha256(attempt["prompt_sha256"], "review prompt_sha256") != prompt_sha256: + raise ContractError("review prompt hash does not match the frozen prompt") + review_sha256 = hashlib.sha256(canonical_json(review).encode()).hexdigest() + if _sha256(attempt["review_sha256"], "review document sha256") != review_sha256: + raise ContractError("review document hash is invalid") + if attempt["worker_invocations"] != 1 or type(attempt["worker_invocations"]) is not int: + raise ContractError("review worker invocation accounting is invalid") + if attempt["worker_invocations"] > task["budget"]["max_worker_invocations"]: + raise ContractError("review exceeds max_worker_invocations") + native_requests = attempt["native_model_requests"] + if native_requests is not None and ( + type(native_requests) is not int or native_requests < 0): + raise ContractError("native model request count must be non-negative or unknown") + usage = _exact(attempt["usage"], { + "input_tokens", "output_tokens", "total_tokens", "source", + }, "review usage") + for field in ("input_tokens", "output_tokens", "total_tokens"): + if usage[field] is not None and ( + type(usage[field]) is not int or usage[field] < 0): + raise ContractError("review token usage must be non-negative or unknown") + if usage["source"] not in {"native_reported", "unavailable"}: + raise ContractError("review usage source is invalid") + if usage["source"] == "unavailable" and any( + usage[field] is not None + for field in ("input_tokens", "output_tokens", "total_tokens") + ): + raise ContractError("unavailable usage cannot invent token counts") + return json.loads(canonical_json(document)) + + +def make_branch_review_evidence( + snapshot: dict[str, Any], + review: dict[str, Any], + checks: list[dict[str, Any]], +) -> dict[str, Any]: + task, workspace = snapshot["task"], snapshot["workspace"] + normalized_review = validate_review_document(review, task, workspace) + normalized_checks = validate_check_results(checks, task, workspace) + candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + document = { + "schema_version": 1, + "workflow": "branch-review", + "candidate_sha256": candidate_sha256, + "base_oid": base_oid, + "target_oid": target_oid, + "review": normalized_review, + "checks": normalized_checks, + "evaluation": evaluate_branch_review( + task, workspace, normalized_review, normalized_checks, + ), + "attempt": { + "role": "reviewer", + "selected_profile": snapshot["routing"]["roles"]["reviewer"]["selected"], + "prompt_sha256": hashlib.sha256( + build_review_prompt(task, workspace).encode() + ).hexdigest(), + "review_sha256": hashlib.sha256( + canonical_json(normalized_review).encode() + ).hexdigest(), + "worker_invocations": 1, + "native_model_requests": None, + "usage": { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + }, + } + return validate_branch_review_evidence(document, snapshot) + + +def decode_branch_review_evidence( + payload: bytes | str, + snapshot: dict[str, Any], +) -> dict[str, Any]: + return validate_branch_review_evidence( + _strict_json_object( + payload, "branch review evidence", maximum=MAX_EVIDENCE_BYTES, + ), + snapshot, + ) diff --git a/plugin/core/src/devsquad/workspaces.py b/plugin/core/src/devsquad/workspaces.py index b82be38..e062816 100644 --- a/plugin/core/src/devsquad/workspaces.py +++ b/plugin/core/src/devsquad/workspaces.py @@ -189,25 +189,21 @@ def _validate_workspace( raise ContractError(f"scoped symlink escapes the review workspace: {name}") -def prepare_review_workspace( +def _prepare_detached_workspace( source_repo: Path, runtime: Path, project_id: str, run_id: str, - base_oid: str, target_oid: str, scope_paths: Iterable[str], - *, - required_clean_paths: Iterable[str] = (), -) -> dict[str, object]: - """Create or validate one detached, run-owned worktree at the target commit.""" + name: str, +) -> tuple[Path, tuple[str, ...]]: repo = source_repo.resolve(strict=True) scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) - assert_clean_inputs(repo, scopes, required_clean_paths) project = _validate_segment(project_id, "project id") run = _validate_segment(run_id, "run id") workspace = ( - runtime.resolve() / "projects" / project / "runs" / run / "review-worktree" + runtime.resolve() / "projects" / project / "runs" / run / name ) workspace.parent.mkdir(parents=True, exist_ok=True) if not workspace.exists(): @@ -225,8 +221,30 @@ def prepare_review_workspace( ) from exc if result.returncode != 0 and not workspace.exists(): detail = result.stderr.decode("utf-8", "replace").strip() - raise ContractError(f"could not create frozen review workspace: {detail}") + label = name.replace("-", " ") + raise ContractError(f"could not create frozen {label}: {detail}") _validate_workspace(repo, workspace, target_oid, scopes) + return workspace.resolve(), scopes + + +def prepare_review_workspace( + source_repo: Path, + runtime: Path, + project_id: str, + run_id: str, + base_oid: str, + target_oid: str, + scope_paths: Iterable[str], + *, + required_clean_paths: Iterable[str] = (), +) -> dict[str, object]: + """Create or validate one detached, run-owned worktree at the target commit.""" + repo = source_repo.resolve(strict=True) + scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) + assert_clean_inputs(repo, scopes, required_clean_paths) + workspace, scopes = _prepare_detached_workspace( + repo, runtime, project_id, run_id, target_oid, scopes, "review-worktree", + ) changed = _decode_paths( _git( repo, "diff", "--no-renames", "--name-only", "-z", @@ -246,3 +264,28 @@ def prepare_review_workspace( "candidate_sha256": hashlib.sha256(canonical_json(identity).encode()).hexdigest(), "scope": list(scopes), } + + +def prepare_check_workspace( + source_repo: Path, + runtime: Path, + project_id: str, + run_id: str, + target_oid: str, + scope_paths: Iterable[str], + *, + required_clean_paths: Iterable[str] = (), +) -> dict[str, object]: + """Create an independent candidate worktree for trusted declared checks.""" + repo = source_repo.resolve(strict=True) + scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) + assert_clean_inputs(repo, scopes, required_clean_paths) + workspace, scopes = _prepare_detached_workspace( + repo, runtime, project_id, run_id, target_oid, scopes, "check-worktree", + ) + return { + "schema_version": 1, + "path": str(workspace), + "target_oid": target_oid, + "scope": list(scopes), + } diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 11cde12..20b4a5e 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -21,6 +21,7 @@ Store, request_hash, ) +from devsquad.contracts import ContractError class HandoffStoreTest(unittest.TestCase): @@ -263,6 +264,81 @@ def test_publish_is_fenced_and_atomically_releases_writer(self): with self.assertRaises(ConflictError): self.store.handoff_snapshot(run_id) + def test_durable_evidence_and_handoff_cross_one_atomic_fence(self): + run_id, reservation, _ = self.running_run("durable-handoff") + artifacts = [] + for name, content in ( + (f"{reservation.attempt_id}.stdout", b'{"workflow":"review"}\n'), + (f"{reservation.attempt_id}.stderr", b""), + ("review.json", b'{"verdict":"clean"}\n'), + ): + path, digest, size = self.store.finalize_artifact(run_id, name, content) + artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + review = next(item for item in artifacts if item["name"] == "review.json") + packet = { + "schema_version": 1, + "candidate_sha256": "c" * 64, + "artifacts": [{"name": review["name"], "sha256": review["sha256"]}], + } + outcome = self.store.commit_durable_handoff( + run_id, + reservation.attempt_token, + artifacts, + {"stdout": {"captured_bytes": 22}, "stderr": {"captured_bytes": 0}}, + packet, + ) + self.assertEqual(outcome, "awaiting_host") + snapshot = self.store.handoff_snapshot(run_id) + self.assertEqual(snapshot.packet["candidate_sha256"], "c" * 64) + reference = snapshot.packet["artifacts"][0] + self.assertEqual(set(reference), {"artifact_id", "name", "sha256"}) + artifact = self.store.connection.execute( + "SELECT name,sha256 FROM artifacts WHERE id=?", (reference["artifact_id"],), + ).fetchone() + self.assertEqual((artifact["name"], artifact["sha256"]), ( + "review.json", review["sha256"], + )) + self.assertEqual( + self.store.commit_durable_handoff( + run_id, + reservation.attempt_token, + artifacts, + {"stdout": {"captured_bytes": 22}, "stderr": {"captured_bytes": 0}}, + packet, + ), + "awaiting_host", + ) + + invalid_run, invalid_reservation, _ = self.running_run("invalid-evidence") + invalid_artifacts = [] + for name, content in ( + (f"{invalid_reservation.attempt_id}.stdout", b"output"), + (f"{invalid_reservation.attempt_id}.stderr", b""), + ): + path, digest, size = self.store.finalize_artifact( + invalid_run, name, content, + ) + invalid_artifacts.append({ + "name": name, "path": path, "sha256": digest, "byte_size": size, + }) + with self.assertRaisesRegex(ContractError, "does not match durable evidence"): + self.store.commit_durable_handoff( + invalid_run, + invalid_reservation.attempt_token, + invalid_artifacts, + {}, + { + "schema_version": 1, + "artifacts": [{"name": "missing.json", "sha256": "0" * 64}], + }, + ) + self.assertEqual(self.store.run(invalid_run)["state"], "running") + def test_claim_cas_renewal_and_expired_takeover(self): run_id, _, snapshot = self.waiting_run() first_time = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py new file mode 100644 index 0000000..bfa6bdd --- /dev/null +++ b/test/core/test_review_runtime.py @@ -0,0 +1,188 @@ +from __future__ import annotations + +import hashlib +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import time +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.service import Service +from devsquad.store import Store +from devsquad_test_fixtures import branch_review_routing_documents + + +class DurableBranchReviewTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-review-runtime-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], + check=True, + ) + (self.repo / "src").mkdir() + (self.repo / "tests").mkdir() + (self.repo / "devsquad").mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# fixture\n") + profiles, policy = branch_review_routing_documents() + (self.repo / "devsquad/profiles.json").write_text(profiles) + (self.repo / "devsquad/policy.json").write_text(policy) + subprocess.run(["git", "-C", str(self.repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.base = self.git_text("rev-parse", "HEAD").strip() + (self.repo / "src/app.py").write_text("VALUE = 'candidate'\n") + subprocess.run(["git", "-C", str(self.repo), "add", "src/app.py"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "candidate"], + check=True, + ) + self.target = self.git_text("rev-parse", "HEAD").strip() + self.task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + self.task["project"] = { + "repo_path": str(self.repo), + "base_ref": self.base, + "target_ref": self.target, + } + self.task["checks"] = [{ + "id": "fixture-tests", + "argv": [sys.executable, "-c", "print('reported failure'); raise SystemExit(7)"], + "cwd": ".", + "timeout_seconds": 10, + "required_to_pass": False, + }] + self.service = Service(self.runtime) + self.fixture = { + "verdict": "findings", + "summary": "The candidate changes the configured value.", + "findings": [{ + "id": "F-1", + "severity": "medium", + "title": "Changed behavior needs confirmation", + "description": "The new value differs from the baseline.", + "path": "src/app.py", + "start_line": 1, + "end_line": 1, + "evidence": "The target contains VALUE = 'candidate'.", + }], + } + + def git_bytes(self, *arguments): + return subprocess.run( + ["git", "-C", str(self.repo), *arguments], + check=True, + capture_output=True, + ).stdout + + def git_text(self, *arguments): + return self.git_bytes(*arguments).decode() + + def wait_state(self, run_id, states, timeout=10): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] in states: + return status + time.sleep(0.05) + log = self.runtime / "private-logs" / f"{run_id}.supervisor.log" + detail = log.read_text() if log.exists() else "no supervisor log" + self.fail(f"run did not reach {states}: {self.service.status(run_id)}\n{detail}") + + def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self): + (self.repo / "notes.txt").write_text("unrelated local work\n") + before_head = self.git_bytes("rev-parse", "HEAD") + before_status = self.git_bytes("status", "--porcelain=v1", "-z") + before_index = hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest() + + started = self.service.start( + self.task, + "durable-review", + _internal_review_fixture=self.fixture, + ) + self.assertTrue(started["created"]) + waiting = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + self.assertEqual(waiting["next_action"], "claim_handoff") + self.assertEqual( + (self.git_bytes("rev-parse", "HEAD"), + self.git_bytes("status", "--porcelain=v1", "-z"), + hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest()), + (before_head, before_status, before_index), + ) + + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + run = store.run(started["run_id"]) + snapshot = json.loads(run["mutable_snapshot"]) + self.assertEqual(run["worktree_path"], snapshot["workspace"]["path"]) + self.assertNotEqual( + snapshot["workspace"]["path"], snapshot["check_workspace"]["path"], + ) + attempts = store.connection.execute( + "SELECT status FROM attempts WHERE run_id=?", (started["run_id"],), + ).fetchall() + self.assertEqual([row["status"] for row in attempts], ["finished"]) + artifacts = store.result_snapshot(started["run_id"])[1] + names = {item["name"] for item in artifacts} + self.assertTrue({ + "review.json", "checks.json", "evaluation.json", "review-attempt.json", + } <= names) + self.assertNotIn("result-receipt.json", names) + handoff = store.handoff_snapshot(started["run_id"]) + finally: + store.close() + + packet = handoff.packet + self.assertEqual(packet["candidate_sha256"], snapshot["workspace"]["candidate_sha256"]) + self.assertEqual(packet["review"]["findings"][0]["id"], "F-1") + self.assertEqual(packet["checks"][0]["status"], "failed") + self.assertFalse(packet["checks"][0]["required_to_pass"]) + self.assertTrue(packet["evaluation"]["accept_allowed"]) + self.assertEqual(packet["evaluation"]["report_only_failures"], ["fixture-tests"]) + self.assertEqual( + {reference["name"] for reference in packet["artifacts"]}, + {"review.json", "checks.json", "evaluation.json", "review-attempt.json"}, + ) + self.assertTrue(all(reference["artifact_id"] for reference in packet["artifacts"])) + + claimed = self.service.handoff_claim( + started["run_id"], waiting["version"], "host-terminal", + ) + self.assertEqual(claimed["handoff"]["packet"], packet) + self.assertFalse(self.service.result(started["run_id"])["ready"]) + + def test_invalid_internal_review_fails_before_launch_with_a_receipt(self): + invalid = dict(self.fixture, verdict="clean") + started = self.service.start( + self.task, + "invalid-review", + _internal_review_fixture=invalid, + ) + self.assertEqual(started["state"], "failed") + self.assertEqual(started["error"]["error"], "PREPARATION_FAILED") + self.assertIn("verdict and findings disagree", started["error"]["message"]) + result = self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + self.assertEqual( + [artifact["name"] for artifact in result["artifacts"]], + ["result-receipt.json"], + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_review_workflow.py b/test/core/test_review_workflow.py index b35bacd..8ec3c3f 100644 --- a/test/core/test_review_workflow.py +++ b/test/core/test_review_workflow.py @@ -14,6 +14,8 @@ build_review_prompt, decode_review_document, evaluate_branch_review, + make_branch_review_evidence, + validate_branch_review_evidence, validate_check_results, validate_review_document, ) @@ -50,6 +52,23 @@ def setUp(self): }], } self.check = self.check_result("failed", returncode=1) + self.snapshot = { + "task": self.task, + "workspace": self.workspace, + "routing": { + "roles": { + "reviewer": { + "selected": { + "profile_id": "fixture-reviewer", + "profile_sha256": "1" * 64, + "profile": {"harness": "fixture"}, + "reference": {"kind": "profile", "id": "fixture-reviewer"}, + "binding": None, + }, + }, + }, + }, + } def stream(self, content=""): encoded = content.encode() @@ -243,6 +262,29 @@ def test_prompt_is_deterministic_read_only_and_distinguishes_adversarial_mode(se self.assertIn('"focus":"trust boundaries"', prompt) self.assertNotEqual(prompt, first) + def test_combined_evidence_recomputes_gates_profile_and_accounting(self): + evidence = make_branch_review_evidence( + self.snapshot, self.review, [self.check], + ) + self.assertEqual( + validate_branch_review_evidence(evidence, self.snapshot), evidence, + ) + for mutate, message in ( + (lambda value: value["evaluation"].__setitem__("accept_allowed", False), + "derived gates"), + (lambda value: value["attempt"].__setitem__( + "selected_profile", {"profile_id": "substituted"}), + "frozen selected profile"), + (lambda value: value["attempt"].__setitem__("worker_invocations", 2), + "invocation accounting"), + (lambda value: value["attempt"]["usage"].__setitem__("total_tokens", 0), + "cannot invent token counts"), + ): + invalid = copy.deepcopy(evidence) + mutate(invalid) + with self.assertRaisesRegex(ContractError, message): + validate_branch_review_evidence(invalid, self.snapshot) + if __name__ == "__main__": unittest.main() diff --git a/test/core/test_service.py b/test/core/test_service.py index ec9596e..29316fb 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -565,7 +565,11 @@ def test_public_preflight_freezes_profile_policy_and_alias_selection(self): hashlib.sha256(canonical_json(identity).encode()).hexdigest(), ) frozen=Path(workspace["path"]) + check_workspace=Path(snapshot["check_workspace"]["path"]) + self.assertNotEqual(check_workspace,frozen) + self.assertEqual(snapshot["check_workspace"]["target_oid"],target_oid) self.assertEqual((frozen/"src/app.py").read_text(),"VALUE = 'candidate'\n") + self.assertEqual((check_workspace/"src/app.py").read_text(),"VALUE = 'candidate'\n") self.assertEqual( subprocess.run( ["git","-C",str(frozen),"rev-parse","--abbrev-ref","HEAD"], diff --git a/test/core/test_validation.py b/test/core/test_validation.py index 5492ecd..46e6ca1 100644 --- a/test/core/test_validation.py +++ b/test/core/test_validation.py @@ -51,6 +51,15 @@ def test_task_rejects_adversarial_nested_types(self): lambda t: t["origin"].__setitem__("session_ref", []), lambda t: t["acceptance"].append(copy.deepcopy(t["acceptance"][0])), lambda t: t["checks"].append(copy.deepcopy(t["checks"][0])), + lambda t: t.__setitem__( + "acceptance", [copy.deepcopy(t["acceptance"][0])] * 101, + ), + lambda t: t.__setitem__( + "checks", [copy.deepcopy(t["checks"][0])] * 17, + ), + lambda t: t["scope"].__setitem__( + "read_paths", [f"path-{i}" for i in range(257)], + ), ] for mutate in mutations: value = copy.deepcopy(self.task); mutate(value) diff --git a/test/core/test_workspaces.py b/test/core/test_workspaces.py index ae5f889..766a901 100644 --- a/test/core/test_workspaces.py +++ b/test/core/test_workspaces.py @@ -12,6 +12,7 @@ from devsquad.workspaces import ( assert_clean_inputs, committed_regular_file, + prepare_check_workspace, prepare_review_workspace, repo_relative_config, resolve_commit, @@ -104,6 +105,27 @@ def test_workspace_is_detached_frozen_idempotent_and_checkout_preserving(self): self.assertEqual(second, first) self.assertEqual((workspace / "src/app.py").read_text(), "VALUE = 'candidate'\n") + def test_checks_get_a_separate_detached_candidate_workspace(self): + review = self.prepare() + check = prepare_check_workspace( + self.repo, + self.runtime, + "project-1", + "run-1", + self.target, + ("src", "tests"), + required_clean_paths=("devsquad/profiles.json", "devsquad/policy.json"), + ) + review_path, check_path = Path(review["path"]), Path(check["path"]) + self.assertNotEqual(review_path, check_path) + self.assertEqual((check_path / "src/app.py").read_text(), "VALUE = 'candidate'\n") + self.assertEqual( + self.git("-C", str(check_path), "rev-parse", "--abbrev-ref", "HEAD").strip(), + b"HEAD", + ) + (check_path / "src/app.py").write_text("check side effect\n") + self.assertEqual((review_path / "src/app.py").read_text(), "VALUE = 'candidate'\n") + def test_dirty_scoped_tracked_path_is_rejected(self): (self.repo / "src/app.py").write_text("dirty\n") with self.assertRaisesRegex(ContractError, "src/app.py"): From 30cf49e1a306974adc6ddf0a2e9799c45d794e03 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 17 Sep 2026 16:35:24 +0530 Subject: [PATCH 053/197] feat: complete M3 host review dispositions --- plugin/core/src/devsquad/reports.py | 334 +++++++++++++++++++++ plugin/core/src/devsquad/review_worker.py | 8 +- plugin/core/src/devsquad/service.py | 255 +++++++++++++++- plugin/core/src/devsquad/store.py | 350 ++++++++++++++++++++++ plugin/core/src/devsquad/supervisor.py | 10 +- plugin/core/src/devsquad/workflows.py | 69 +++++ plugin/core/src/devsquad/workspaces.py | 30 +- test/core/test_review_runtime.py | 216 ++++++++++++- 8 files changed, 1256 insertions(+), 16 deletions(-) create mode 100644 plugin/core/src/devsquad/reports.py diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py new file mode 100644 index 0000000..a5d60cb --- /dev/null +++ b/plugin/core/src/devsquad/reports.py @@ -0,0 +1,334 @@ +"""Deterministic M3 terminal receipt, Markdown, event-export and manifest builders.""" + +from __future__ import annotations + +import hashlib +from typing import Any + +from .contracts import ContractError +from .store import canonical_json, request_hash +from .workflows import ( + validate_branch_review_handoff, + validate_handoff_decision_evidence, +) + + +TERMINAL_REPORT_NAMES = frozenset({ + "receipt.json", + "receipt.md", + "events.jsonl", + "artifact-manifest.json", + "result-receipt.json", +}) + + +def _artifact_projection(artifact: dict[str, Any]) -> dict[str, Any]: + required = {"id", "name", "sha256", "byte_size"} + if not isinstance(artifact, dict) or not required <= set(artifact): + raise ContractError("report artifact projection is invalid") + if (not isinstance(artifact["id"], str) or not artifact["id"] + or not isinstance(artifact["name"], str) or not artifact["name"] + or not isinstance(artifact["sha256"], str) + or len(artifact["sha256"]) != 64 + or any(character not in "0123456789abcdef" + for character in artifact["sha256"]) + or type(artifact["byte_size"]) is not int + or artifact["byte_size"] < 0): + raise ContractError("report artifact projection values are invalid") + return {key: artifact[key] for key in ("id", "name", "sha256", "byte_size")} + + +def _event_export(events: list[dict[str, Any]]) -> tuple[bytes, int, int]: + if not isinstance(events, list): + raise ContractError("events export input must be an array") + lines = [] + last_cursor = 0 + last_version = 0 + for event in events: + if not isinstance(event, dict): + raise ContractError("events export entries must be objects") + cursor, version = event.get("id"), event.get("run_version") + if (type(cursor) is not int or cursor <= last_cursor + or type(version) is not int or version <= last_version): + raise ContractError("events export order is invalid") + last_cursor, last_version = cursor, version + lines.append(canonical_json(event)) + content = (("\n".join(lines) + "\n") if lines else "").encode() + return content, last_cursor, last_version + + +def _decision(value: Any) -> dict[str, Any]: + fields = { + "schema_version", "submission_id", "submission_hash", "disposition", + "reason", "evidence_refs", + } + if not isinstance(value, dict) or set(value) != fields: + raise ContractError("report lead decision fields are invalid") + if value["schema_version"] != 1 or type(value["schema_version"]) is not int: + raise ContractError("report lead decision schema_version is invalid") + if (not isinstance(value["submission_id"], str) or not value["submission_id"] + or value["disposition"] not in {"accept", "revise", "reject"} + or not isinstance(value["reason"], str) + or (value["disposition"] == "revise" and not value["reason"])): + raise ContractError("report lead decision values are invalid") + body = {key: value[key] for key in fields if key != "submission_hash"} + if value["submission_hash"] != request_hash(body): + raise ContractError("report lead decision hash is invalid") + return value + + +def _history( + entries: list[dict[str, Any]], + snapshot: dict[str, Any], +) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + if not isinstance(entries, list) or not entries: + raise ContractError("branch review report history must be non-empty") + attempts = [] + dispositions = [] + prior_sequence = 0 + candidate = None + for index, entry in enumerate(entries): + required = { + "handoff_id", "sequence", "packet", "packet_sha256", "decision", + "recorded_run_version", + } + if not isinstance(entry, dict) or set(entry) != required: + raise ContractError("branch review report history entry is invalid") + if (not isinstance(entry["handoff_id"], str) or not entry["handoff_id"] + or type(entry["sequence"]) is not int + or entry["sequence"] <= prior_sequence + or type(entry["recorded_run_version"]) is not int + or entry["recorded_run_version"] < 1): + raise ContractError("branch review report history order is invalid") + packet_json = canonical_json(entry["packet"]) + if hashlib.sha256(packet_json.encode()).hexdigest() != entry["packet_sha256"]: + raise ContractError("branch review report handoff hash is invalid") + packet = validate_branch_review_handoff(entry["packet"], snapshot) + decision = _decision(entry["decision"]) + validate_handoff_decision_evidence(decision, packet) + identity = ( + packet["candidate_sha256"], packet["base_oid"], packet["target_oid"], + ) + if candidate is None: + candidate = identity + elif identity != candidate: + raise ContractError("branch review report history changes the candidate") + if index < len(entries) - 1 and decision["disposition"] != "revise": + raise ContractError("only a revision may precede another review attempt") + attempts.append({ + "id": packet["attempt_id"], + "sequence": entry["sequence"], + **packet["attempt"], + "review": packet["review"], + "checks": packet["checks"], + "evaluation": packet["evaluation"], + "evidence_refs": packet["artifacts"], + }) + dispositions.append({ + "handoff_id": entry["handoff_id"], + "sequence": entry["sequence"], + "submission_id": decision["submission_id"], + "submission_hash": decision["submission_hash"], + "disposition": decision["disposition"], + "reason": decision["reason"], + "evidence_refs": decision["evidence_refs"], + "recorded_run_version": entry["recorded_run_version"], + }) + prior_sequence = entry["sequence"] + return attempts, dispositions + + +def _markdown(receipt: dict[str, Any]) -> str: + review = receipt["review"] + lines = [ + "# DevSquad branch review", + "", + f"- Run: `{receipt['run_id']}`", + f"- State: `{receipt['state']}`", + f"- Disposition: `{receipt['lead']['disposition']}`", + f"- Candidate: `{receipt['candidate']['sha256']}`", + f"- Base: `{receipt['candidate']['base_oid']}`", + f"- Target: `{receipt['candidate']['target_oid']}`", + f"- Review verdict: `{review['verdict']}`", + f"- Review attempts: `{len(receipt['attempts'])}`", + "", + "## Review", + "", + review["summary"], + "", + ] + if review["findings"]: + lines.extend(["## Findings", ""]) + for finding in review["findings"]: + lines.extend([ + f"- **{finding['severity'].upper()} — {finding['title']}** " + f"(`{finding['path']}:{finding['start_line']}`): " + f"{finding['description']}", + f" Evidence: {finding['evidence']}", + ]) + lines.append("") + lines.extend(["## Checks", ""]) + if receipt["checks"]: + for check in receipt["checks"]: + requirement = "required" if check["required_to_pass"] else "report-only" + lines.append(f"- `{check['id']}`: **{check['status']}** ({requirement})") + else: + lines.append("- No checks were declared.") + lines.extend([ + "", + "## Lead disposition", + "", + receipt["lead"]["reason"] or "No reason supplied.", + "", + "## Limitations", + "", + ]) + if receipt["limitations"]: + lines.extend(f"- {item}" for item in receipt["limitations"]) + else: + lines.append("- None recorded.") + return "\n".join(lines) + "\n" + + +def build_terminal_reports( + *, + run_id: str, + state: str, + snapshot: dict[str, Any], + history: list[dict[str, Any]], + run_artifacts: list[dict[str, Any]], + events: list[dict[str, Any]], + completed_at: str, + error: dict[str, Any] | None = None, +) -> dict[str, bytes]: + if not isinstance(run_id, str) or not run_id: + raise ContractError("report run id is invalid") + if state not in {"succeeded", "failed"}: + raise ContractError("branch review report state is invalid") + if not isinstance(snapshot, dict) or not isinstance(snapshot.get("routing"), dict): + raise ContractError("branch review report snapshot is invalid") + attempts, dispositions = _history(history, snapshot) + final_packet = validate_branch_review_handoff(history[-1]["packet"], snapshot) + final_decision = _decision(history[-1]["decision"]) + if (state == "succeeded") != (final_decision["disposition"] == "accept"): + raise ContractError("terminal state and lead disposition disagree") + if not isinstance(completed_at, str) or not completed_at: + raise ContractError("report completion timestamp is invalid") + if error is not None and not isinstance(error, dict): + raise ContractError("report error must be an object or null") + + projected = [_artifact_projection(artifact) for artifact in run_artifacts] + artifact_by_id = {artifact["id"]: artifact for artifact in projected} + if len(artifact_by_id) != len(projected): + raise ContractError("report evidence artifacts are duplicated") + expected_ids = set() + for entry in history: + for reference in entry["packet"]["artifacts"]: + artifact = artifact_by_id.get(reference["artifact_id"]) + if (artifact is None or artifact["name"] != reference["name"] + or artifact["sha256"] != reference["sha256"]): + raise ContractError( + "handoff evidence does not match stored report artifacts" + ) + expected_ids.add(reference["artifact_id"]) + evidence_projected = [ + artifact for artifact in projected if artifact["id"] in expected_ids + ] + + events_content, through_cursor, through_version = _event_export(events) + limitations = [] + if "internal_review_fixture" in snapshot: + limitations.append( + "Reviewer output came from the explicit offline fixture; " + "it is not live-provider evidence." + ) + native_counts = [attempt["native_model_requests"] for attempt in attempts] + receipt = { + "schema_version": 1, + "run_id": run_id, + "workflow": "branch-review", + "state": state, + "completed_at": completed_at, + "candidate": { + "sha256": final_packet["candidate_sha256"], + "base_oid": final_packet["base_oid"], + "target_oid": final_packet["target_oid"], + }, + "routing": snapshot["routing"], + "review": final_packet["review"], + "checks": final_packet["checks"], + "evaluation": final_packet["evaluation"], + "criteria": final_packet["evaluation"]["criteria"], + "attempts": attempts, + "dispositions": dispositions, + "revisions": { + "requested": sum( + item["disposition"] == "revise" for item in dispositions + ), + "executed": len(attempts) - 1, + "maximum": snapshot["task"]["budget"]["max_revisions"], + }, + "lead": { + "mode": "host", + "disposition": final_decision["disposition"], + "reason": final_decision["reason"], + "submission_id": final_decision["submission_id"], + "submission_hash": final_decision["submission_hash"], + "evidence_refs": final_decision["evidence_refs"], + "usage": { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + }, + "accounting": { + "worker_invocations": sum( + attempt["worker_invocations"] for attempt in attempts + ), + "native_model_requests": ( + None if any(value is None for value in native_counts) + else sum(native_counts) + ), + "attempt_usage": [attempt["usage"] for attempt in attempts], + "host_usage_measured": False, + }, + "artifacts": projected, + "evidence_artifacts": evidence_projected, + "events_export": { + "through_cursor": through_cursor, + "through_run_version": through_version, + "includes_terminal_event": False, + "excludes_terminal_report_artifact_events": True, + }, + "limitations": limitations, + "error": error, + } + receipt_content = (canonical_json(receipt) + "\n").encode() + contents = { + "receipt.json": receipt_content, + "receipt.md": _markdown(receipt).encode(), + "events.jsonl": events_content, + "result-receipt.json": receipt_content, + } + manifest_entries = list(projected) + for name, content in contents.items(): + manifest_entries.append({ + "id": None, + "name": name, + "sha256": hashlib.sha256(content).hexdigest(), + "byte_size": len(content), + }) + manifest = { + "schema_version": 1, + "run_id": run_id, + "candidate_sha256": final_packet["candidate_sha256"], + "artifacts": manifest_entries, + "self_excluded": True, + } + contents["artifact-manifest.json"] = ( + canonical_json(manifest) + "\n" + ).encode() + if set(contents) != TERMINAL_REPORT_NAMES: + raise ContractError("terminal report set is incomplete") + return contents diff --git a/plugin/core/src/devsquad/review_worker.py b/plugin/core/src/devsquad/review_worker.py index 217d392..eed3e34 100644 --- a/plugin/core/src/devsquad/review_worker.py +++ b/plugin/core/src/devsquad/review_worker.py @@ -15,7 +15,7 @@ from .store import canonical_json from .supervisor import BoundedDrain from .workflows import MAX_PREVIEW_CHARS, make_branch_review_evidence -from .workspaces import dirty_paths +from .workspaces import dirty_paths, reset_check_workspace MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 @@ -146,6 +146,12 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: raise ContractError("offline review snapshot is incomplete") review_root = Path(workspace["path"]).resolve(strict=True) checks_root = Path(check_workspace["path"]).resolve(strict=True) + reset_check_workspace( + review_root, + checks_root, + workspace["target_oid"], + check_workspace["scope"], + ) if dirty_paths(review_root): raise ContractError("frozen review workspace is dirty before reviewer execution") review = fixture diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 043b72c..b14e152 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -4,7 +4,7 @@ import hashlib import json import os -from datetime import datetime +from datetime import datetime, timezone from pathlib import Path import shutil import subprocess @@ -14,6 +14,7 @@ from typing import Any from .contracts import ContractError +from .reports import build_terminal_reports from .router import load_routing from .store import ( ConflictError, @@ -24,7 +25,13 @@ canonical_json, ) from .validation import validate_task -from .workflows import review_mode, validate_review_document +from .workflows import ( + apply_lead_disposition, + review_mode, + validate_branch_review_handoff, + validate_handoff_decision_evidence, + validate_review_document, +) from .workspaces import ( assert_clean_inputs, committed_regular_file, @@ -460,6 +467,191 @@ def handoff_claim( finally: store.close() + @staticmethod + def _review_snapshot(run: dict[str, Any]) -> dict[str, Any]: + try: + snapshot = json.loads(run["mutable_snapshot"]) + except (KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError("frozen branch review snapshot is invalid") from exc + if (not isinstance(snapshot, dict) + or canonical_json(snapshot) != run["mutable_snapshot"]): + raise ConflictError("frozen branch review snapshot is not canonical") + return snapshot + + @staticmethod + def _review_gate( + store: Store, + run_id: str, + handoff: HandoffSnapshot, + snapshot: dict[str, Any], + decision: dict[str, Any], + ) -> tuple[dict[str, Any], dict[str, Any]]: + packet = validate_branch_review_handoff(handoff.packet, snapshot) + validate_handoff_decision_evidence(decision, packet) + revisions_used = sum( + entry["decision"]["disposition"] == "revise" + for entry in store.branch_review_history(run_id) + if entry["sequence"] < handoff.sequence + ) + gate = apply_lead_disposition( + packet["evaluation"], + decision.get("disposition"), + revisions_used=revisions_used, + max_revisions=snapshot["task"]["budget"]["max_revisions"], + ) + return packet, gate + + def _terminalize_branch_review( + self, + store: Store, + run_id: str, + handoff: HandoffSnapshot, + snapshot: dict[str, Any], + entry: dict[str, Any], + terminal_state: str, + error: dict[str, Any] | None, + ) -> dict[str, Any]: + history = store.branch_review_history(run_id) + evidence_ids = { + reference["artifact_id"] + for item in history + for reference in item["packet"]["artifacts"] + } + run_artifacts = store.artifacts_for_run(run_id) + if evidence_ids - {artifact["id"] for artifact in run_artifacts}: + raise ConflictError("branch review evidence artifact is missing") + expected_parent = (self.artifacts / run_id).resolve() + for artifact in run_artifacts: + path = Path(artifact["path"]) + if (path.resolve().parent != expected_parent + or not path.is_file() + or path.stat().st_size != artifact["byte_size"] + or hashlib.sha256(path.read_bytes()).hexdigest() + != artifact["sha256"]): + raise ConflictError("branch review artifact is missing or corrupt") + reports = build_terminal_reports( + run_id=run_id, + state=terminal_state, + snapshot=snapshot, + history=history, + run_artifacts=run_artifacts, + events=store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + error=error, + ) + prepared = [] + for name in sorted(reports): + content = reports[name] + path, digest, size = store.finalize_artifact(run_id, name, content) + prepared.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + decision = entry["decision"] + outcome = store.complete_handoff_terminal( + run_id, + handoff.handoff_id, + decision["submission_id"], + decision["submission_hash"], + prepared, + terminal_state, + { + "handoff_id": handoff.handoff_id, + "submission_id": decision["submission_id"], + "submission_hash": decision["submission_hash"], + "disposition": decision["disposition"], + "receipt": "result-receipt.json", + "error": error, + }, + ) + return { + "action": "terminal", + "state": outcome["state"], + "version": outcome["version"], + "replayed_continuation": outcome["replayed"], + "launch": None, + } + + def _continue_branch_review_submission( + self, + store: Store, + run_id: str, + handoff: HandoffSnapshot, + snapshot: dict[str, Any], + entry: dict[str, Any], + ) -> dict[str, Any]: + run = store.run(run_id) + if run["state"] in TERMINAL_STATES: + return { + "action": "already_terminal", + "state": run["state"], + "version": run["version"], + "replayed_continuation": True, + "launch": None, + } + _, gate = self._review_gate( + store, run_id, handoff, snapshot, entry["decision"], + ) + if gate["action"] == "repeat_review": + decision = entry["decision"] + requeue = store.requeue_review_revision( + run_id, + handoff.handoff_id, + decision["submission_id"], + decision["submission_hash"], + ) + if requeue["action"] == "requeued": + queued = store.run(run_id) + package, digest = self._verified_package(queued) + return { + "action": "requeued", + "state": queued["state"], + "version": queued["version"], + "replayed_continuation": requeue["replayed"], + "launch": (queued["version"], package, digest), + } + if requeue["action"] == "already_advanced": + advanced = store.run(run_id) + return { + "action": "already_advanced", + "state": advanced["state"], + "version": advanced["version"], + "replayed_continuation": True, + "launch": None, + } + gate = { + **gate, + "action": "budget_exhausted", + "terminal_state": "failed", + } + error = { + "error": "BUDGET_EXHAUSTED", + "message": "branch review worker invocation budget is exhausted", + } + elif gate["action"] == "budget_exhausted": + error = { + "error": "BUDGET_EXHAUSTED", + "message": "branch review revision budget is exhausted", + } + elif gate["terminal_state"] == "failed": + error = { + "error": "REVIEW_REJECTED", + "message": entry["decision"]["reason"] or "host rejected the review", + } + else: + error = None + return self._terminalize_branch_review( + store, + run_id, + handoff, + snapshot, + entry, + gate["terminal_state"], + error, + ) + def handoff_complete( self, run_id: str, @@ -470,10 +662,26 @@ def handoff_complete( if decoded.run_id != run_id: raise ConflictError("handoff completion claim targets a different run") store = self._store() + launch = None try: + run = store.run(run_id) + handoff = store.handoff_snapshot_by_id(run_id, decoded.handoff_id) + branch_review = handoff.packet.get("workflow") == "branch-review" + snapshot = self._review_snapshot(run) if branch_review else None + if branch_review: + self._review_gate(store, run_id, handoff, snapshot, decision) submission = store.record_handoff_submission(run_id, decoded, decision) + continuation = None + if branch_review: + entry = store.recorded_handoff_submission(run_id, decoded.handoff_id) + if entry is None: + raise ConflictError("recorded branch review submission is missing") + continuation = self._continue_branch_review_submission( + store, run_id, handoff, snapshot, entry, + ) + launch = continuation.pop("launch") run = store.run(run_id) - return { + response = { "run_id": run_id, "state": run["state"], "phase": run["phase"], @@ -485,17 +693,47 @@ def handoff_complete( "recorded_run_version": submission.recorded_run_version, "replayed": submission.replayed, } + if continuation is not None: + response["continuation"] = continuation finally: store.close() + if launch is not None: + version, package, digest = launch + self._spawn_daemon(run_id, version, package, digest) + response["launched"] = True + elif branch_review: + response["launched"] = False + return response def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[str, Any]: store = self._store() launch: tuple[int, Path, str] | None = None preparation_error: dict[str, Any] | None = None + branch_response: dict[str, Any] | None = None try: run = store.run(run_id) if run["state"] in TERMINAL_STATES: raise ConflictError("terminal run cannot resume; start a superseding run") - if run["state"] in {"running","cancelling"}: + if run["state"] == "awaiting_host" and run["phase"] == "handoff_submitted": + handoff = store.handoff_snapshot(run_id) + if handoff is None or handoff.packet.get("workflow") != "branch-review": + raise ConflictError("run has no resumable branch review submission") + snapshot = self._review_snapshot(run) + entry = store.recorded_handoff_submission(run_id, handoff.handoff_id) + if entry is None: + raise ConflictError("recorded branch review submission is missing") + continuation = self._continue_branch_review_submission( + store, run_id, handoff, snapshot, entry, + ) + launch = continuation.pop("launch") + current = store.run(run_id) + branch_response = { + "run_id": run_id, + "disposition": continuation["action"], + "state": current["state"], + "version": current["version"], + "launched": launch is not None, + } + if branch_response is None and run["state"] in {"running","cancelling"}: from .supervisor import Supervisor attempt=store.attempt(run_id) disposition = Supervisor(store).import_durable(run_id) if attempt and attempt.get("exit_record") else Supervisor(store).recover(run_id) @@ -503,7 +741,9 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st return {"run_id": run_id, "disposition": disposition, "launched": False} run = store.run(run_id) version = run["version"] - if run["state"] == "queued" and run["phase"] == "preparing": + if branch_response is not None: + pass + elif run["state"] == "queued" and run["phase"] == "preparing": submitted = json.loads(run["submitted_request"]) claim = store.reclaim_preparation( run_id, run["version"], f"preflight-recovery:{os.getpid()}", @@ -528,6 +768,11 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st else: raise ConflictError("run is not resumable") finally: store.close() + if branch_response is not None: + if launch is not None: + version, package, digest = launch + self._spawn_daemon(run_id, version, package, digest) + return branch_response if run["phase"] == "preparing": if launch is None: return { diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 9fd8779..d1adb7a 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -20,6 +20,13 @@ SUPPORTED_SCHEMA_VERSION = 5 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 +BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ + "receipt.json", + "receipt.md", + "events.jsonl", + "artifact-manifest.json", + "result-receipt.json", +}) class ConflictError(ContractError): @@ -742,6 +749,43 @@ def _prepare_durable_artifacts( raise ContractError(f"durable import requires exactly one {requirement}") return prepared, stdout_names[0], stderr_names[0] + def _prepare_exact_artifacts( + self, + run_id: str, + artifacts: list[dict[str, Any]], + expected_names: frozenset[str], + ) -> list[tuple[str, str, str, int]]: + """Verify a complete named artifact set before entering a write transaction.""" + if not isinstance(artifacts, list): + raise ContractError("artifacts must be an array") + expected_parent = (self.artifacts / run_id).resolve() + prepared = [] + names = set() + for artifact in artifacts: + if not isinstance(artifact, dict): + raise ContractError("artifacts must be objects") + name = artifact.get("name") + path = Path(artifact.get("path", "")) + digest = artifact.get("sha256") + size = artifact.get("byte_size") + if (not isinstance(name, str) or not name or Path(name).name != name + or name in names): + raise ContractError("artifact names must be unique path components") + names.add(name) + if path.resolve().parent != expected_parent: + raise ContractError("artifact path is outside the run-owned store") + content = path.read_bytes() + if (not isinstance(digest, str) + or type(size) is not int + or size < 0 + or hashlib.sha256(content).hexdigest() != digest + or len(content) != size): + raise ConflictError("artifact changed before database import") + prepared.append((name, str(path), digest, size)) + if names != set(expected_names): + raise ContractError("terminal artifact set is incomplete or unexpected") + return prepared + def _reference_prepared_artifacts( self, run_id: str, @@ -1431,6 +1475,93 @@ def handoff_snapshot(self, run_id: str) -> HandoffSnapshot | None: self.connection.execute("ROLLBACK") raise + def handoff_snapshot_by_id( + self, run_id: str, handoff_id: str, + ) -> HandoffSnapshot: + if not isinstance(handoff_id, str) or not handoff_id: + raise ContractError("handoff id is required") + self.connection.execute("BEGIN") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if not run: + raise ContractError("run does not exist") + handoff = self.connection.execute( + "SELECT * FROM handoffs WHERE id=? AND run_id=?", (handoff_id, run_id), + ).fetchone() + if not handoff: + raise ConflictError("handoff belongs to a different run") + claim_row = self.connection.execute( + "SELECT run_id,handoff_id,owner_id,fencing_token,lease_expires_at," + "? AS run_version FROM claims WHERE run_id=? AND kind='host' " + "AND handoff_id=? AND active=1", + (run["version"], run_id, handoff_id), + ).fetchone() + snapshot = self._handoff_snapshot_from_rows( + run_id, run, handoff, claim_row, + ) + if snapshot is None: # Defensive: handoff was selected above. + raise ConflictError("handoff is missing") + self.connection.execute("COMMIT") + return snapshot + except Exception: + self.connection.execute("ROLLBACK") + raise + + def branch_review_history(self, run_id: str) -> list[dict[str, Any]]: + """Return canonical recorded handoffs and decisions in workflow order.""" + if not self.connection.execute( + "SELECT 1 FROM runs WHERE id=?", (run_id,), + ).fetchone(): + raise ContractError("run does not exist") + rows = self.connection.execute( + "SELECT h.id AS handoff_id,h.sequence,h.packet_json,h.packet_sha256," + "s.submission_id,s.submission_hash,s.disposition,s.decision_json," + "s.recorded_run_version " + "FROM handoffs h JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE h.run_id=? AND s.outcome='recorded' ORDER BY h.sequence", + (run_id,), + ).fetchall() + history = [] + for row in rows: + try: + packet = json.loads(row["packet_json"]) + decision = json.loads(row["decision_json"]) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("persisted branch review history is invalid") from exc + if (not isinstance(packet, dict) + or canonical_json(packet) != row["packet_json"] + or hashlib.sha256(row["packet_json"].encode()).hexdigest() + != row["packet_sha256"] + or not isinstance(decision, dict) + or canonical_json(decision) != row["decision_json"]): + raise ConflictError("persisted branch review history hash is invalid") + submission_id, submission_hash, disposition, _, _ = ( + self._validated_handoff_decision(run_id, decision) + ) + if (submission_id != row["submission_id"] + or submission_hash != row["submission_hash"] + or disposition != row["disposition"]): + raise ConflictError("persisted branch review decision is inconsistent") + history.append({ + "handoff_id": row["handoff_id"], + "sequence": row["sequence"], + "packet": packet, + "packet_sha256": row["packet_sha256"], + "decision": decision, + "recorded_run_version": row["recorded_run_version"], + }) + return history + + def recorded_handoff_submission( + self, run_id: str, handoff_id: str, + ) -> dict[str, Any] | None: + for entry in self.branch_review_history(run_id): + if entry["handoff_id"] == handoff_id: + return entry + return None + def claim_handoff( self, run_id: str, @@ -1731,6 +1862,204 @@ def record_handoff_submission( raise ConflictError("handoff completion was not recorded") return result + def requeue_review_revision( + self, + run_id: str, + handoff_id: str, + submission_id: str, + submission_hash: str, + ) -> dict[str, Any]: + """Consume one saved revise decision and requeue within both task budgets.""" + if not all( + isinstance(value, str) and value + for value in (handoff_id, submission_id, submission_hash) + ): + raise ContractError("revision requeue identifiers are invalid") + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,r.mutable_snapshot,h.status,h.sequence," + "s.disposition FROM runs r JOIN handoffs h ON h.run_id=r.id " + "JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE r.id=? AND h.id=? AND s.submission_id=? " + "AND s.submission_hash=? AND s.outcome='recorded'", + (run_id, handoff_id, submission_id, submission_hash), + ).fetchone() + if not row: + raise ConflictError("recorded revision submission is missing") + if row["disposition"] != "revise": + raise ConflictError("handoff submission is not a revision request") + latest_sequence = self.connection.execute( + "SELECT MAX(sequence) FROM handoffs WHERE run_id=?", (run_id,), + ).fetchone()[0] + if row["status"] == "consumed": + action = ( + "requeued" + if row["sequence"] == latest_sequence + and row["state"] == "queued" + and row["phase"] is None + else "already_advanced" + ) + self.connection.execute("COMMIT") + return { + "action": action, + "version": row["version"], + "replayed": True, + } + if (row["state"] != "awaiting_host" + or row["phase"] != "handoff_submitted" + or row["status"] != "submitted" + or row["sequence"] != latest_sequence): + raise ConflictError("revision handoff is no longer current") + try: + snapshot = json.loads(row["mutable_snapshot"]) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("frozen review snapshot is invalid") from exc + if (not isinstance(snapshot, dict) + or canonical_json(snapshot) != row["mutable_snapshot"]): + raise ConflictError("frozen review snapshot is not canonical") + try: + budget = snapshot["task"]["budget"] + max_revisions = budget["max_revisions"] + max_invocations = budget["max_worker_invocations"] + except (KeyError, TypeError) as exc: + raise ConflictError("frozen review budget is missing") from exc + if (type(max_revisions) is not int or max_revisions < 0 + or type(max_invocations) is not int or max_invocations < 1): + raise ConflictError("frozen review budget is invalid") + revisions = self.connection.execute( + "SELECT COUNT(*) FROM handoff_submissions s " + "JOIN handoffs h ON h.id=s.handoff_id " + "WHERE h.run_id=? AND s.outcome='recorded' " + "AND s.disposition='revise' AND h.sequence<=?", + (run_id, row["sequence"]), + ).fetchone()[0] + invocations = self.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()[0] + if revisions > max_revisions or invocations >= max_invocations: + self.connection.execute("COMMIT") + return { + "action": "budget_exhausted", + "version": row["version"], + "replayed": False, + "revisions_requested": revisions, + "worker_invocations": invocations, + } + now, version = _utc_now(), row["version"] + 1 + self.connection.execute( + "UPDATE handoffs SET status='consumed',closed_at=? WHERE id=?", + (now, handoff_id), + ) + self.connection.execute( + "UPDATE runs SET state='queued',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff_id, + "submission_id": submission_id, + "revisions_requested": revisions, + "worker_invocations": invocations, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.revision_queued',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return { + "action": "requeued", + "version": version, + "replayed": False, + "revisions_requested": revisions, + "worker_invocations": invocations, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def complete_handoff_terminal( + self, + run_id: str, + handoff_id: str, + submission_id: str, + submission_hash: str, + artifacts: list[dict[str, Any]], + terminal_state: str, + payload: dict[str, Any], + ) -> dict[str, Any]: + """Atomically publish M3 terminal reports and consume the host handoff.""" + if terminal_state not in {"succeeded", "failed"}: + raise ContractError("branch review terminal state is invalid") + if not all( + isinstance(value, str) and value + for value in (handoff_id, submission_id, submission_hash) + ): + raise ContractError("terminal handoff identifiers are invalid") + prepared = self._prepare_exact_artifacts( + run_id, artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS, + ) + encoded_payload = canonical_json(payload) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,h.status,h.sequence,s.disposition " + "FROM runs r JOIN handoffs h ON h.run_id=r.id " + "JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE r.id=? AND h.id=? AND s.submission_id=? " + "AND s.submission_hash=? AND s.outcome='recorded'", + (run_id, handoff_id, submission_id, submission_hash), + ).fetchone() + if not row: + raise ConflictError("recorded terminal submission is missing") + expected_state = "succeeded" if row["disposition"] == "accept" else "failed" + if terminal_state != expected_state: + raise ContractError("terminal state contradicts the lead disposition") + if row["state"] in TERMINAL_STATES: + if row["state"] != terminal_state or row["status"] != "consumed": + raise ConflictError("terminal handoff was completed differently") + self.connection.execute("COMMIT") + return { + "state": row["state"], + "version": row["version"], + "replayed": True, + } + latest_sequence = self.connection.execute( + "SELECT MAX(sequence) FROM handoffs WHERE run_id=?", (run_id,), + ).fetchone()[0] + if (row["state"] != "awaiting_host" + or row["phase"] != "handoff_submitted" + or row["status"] != "submitted" + or row["sequence"] != latest_sequence): + raise ConflictError("terminal handoff is no longer current") + version, _ = self._reference_prepared_artifacts( + run_id, row["version"], prepared, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE handoffs SET status='consumed',closed_at=? WHERE id=?", + (now, handoff_id), + ) + self.connection.execute( + "UPDATE claims SET active=0 WHERE run_id=? AND handoff_id=?", + (run_id, handoff_id), + ) + self.connection.execute( + "UPDATE runs SET state=?,phase=NULL,version=?,updated_at=? WHERE id=?", + (terminal_state, version, now, run_id), + ) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,?,?,?)", + (run_id, version, f"run.{terminal_state}", encoded_payload, now), + ) + self.connection.execute("COMMIT") + return {"state": terminal_state, "version": version, "replayed": False} + except Exception: + self.connection.execute("ROLLBACK") + raise + def cancel_host_wait(self, run_id: str, *, now: datetime | None = None) -> int: self.connection.execute("BEGIN IMMEDIATE") try: @@ -1834,6 +2163,27 @@ def events_page(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str consumed = page[-1]["id"] if page else after return {"events": [{**dict(row), "payload": json.loads(row["payload"])} for row in page], "next_cursor": consumed, "has_more": more} + def events_for_run(self, run_id: str) -> list[dict[str, Any]]: + if not self.connection.execute( + "SELECT 1 FROM runs WHERE id=?", (run_id,), + ).fetchone(): + raise ContractError("run does not exist") + rows = self.connection.execute( + "SELECT id,run_version,type,payload,created_at FROM events " + "WHERE run_id=? ORDER BY id", + (run_id,), + ).fetchall() + events = [] + for row in rows: + try: + payload = json.loads(row["payload"]) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("persisted event payload is invalid") from exc + if canonical_json(payload) != row["payload"]: + raise ConflictError("persisted event payload is not canonical") + events.append({**dict(row), "payload": payload}) + return events + def status_snapshot( self, run_id: str, ) -> tuple[dict[str, Any], dict[str, Any] | None, HandoffSnapshot | None]: diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 3a3efb8..e3b0ac7 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -295,16 +295,17 @@ def _commit_review_handoff( stdout: bytes, ) -> str: evidence = decode_branch_review_evidence(stdout, snapshot) + suffix = attempt["id"] documents = { - "review.json": evidence["review"], - "checks.json": { + f"review-{suffix}.json": evidence["review"], + f"checks-{suffix}.json": { "schema_version": 1, "candidate_sha256": evidence["candidate_sha256"], "target_oid": evidence["target_oid"], "results": evidence["checks"], }, - "evaluation.json": evidence["evaluation"], - "review-attempt.json": evidence["attempt"], + f"evaluation-{suffix}.json": evidence["evaluation"], + f"review-attempt-{suffix}.json": evidence["attempt"], } artifacts = list(stream_artifacts) evidence_references = [] @@ -327,6 +328,7 @@ def _commit_review_handoff( "review": evidence["review"], "checks": evidence["checks"], "evaluation": evidence["evaluation"], + "attempt_id": attempt["id"], "attempt": evidence["attempt"], "artifacts": evidence_references, "instructions": ( diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index cdd4def..549bca5 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -527,3 +527,72 @@ def decode_branch_review_evidence( ), snapshot, ) + + +def validate_branch_review_handoff( + packet: dict[str, Any], + snapshot: dict[str, Any], +) -> dict[str, Any]: + """Validate a saved host packet and reconstruct its trusted evidence.""" + value = _exact(packet, { + "schema_version", "workflow", "candidate_sha256", "base_oid", "target_oid", + "review", "checks", "evaluation", "attempt_id", "attempt", "artifacts", + "instructions", + }, "branch review handoff") + evidence = { + field: value[field] + for field in ( + "schema_version", "workflow", "candidate_sha256", "base_oid", "target_oid", + "review", "checks", "evaluation", "attempt", + ) + } + validate_branch_review_evidence(evidence, snapshot) + attempt_id = _text(value["attempt_id"], "handoff attempt id", maximum=200) + artifacts = value["artifacts"] + if not isinstance(artifacts, list) or len(artifacts) != 4: + raise ContractError("branch review handoff must contain four evidence artifacts") + names, identifiers = set(), set() + for reference in artifacts: + item = _exact( + reference, {"artifact_id", "name", "sha256"}, "handoff artifact reference", + ) + artifact_id = _text(item["artifact_id"], "handoff artifact id", maximum=200) + name = _text(item["name"], "handoff artifact name", maximum=255) + _sha256(item["sha256"], "handoff artifact sha256") + if name in names or artifact_id in identifiers: + raise ContractError("branch review handoff artifact references are duplicated") + names.add(name) + identifiers.add(artifact_id) + expected_names = { + f"review-{attempt_id}.json", + f"checks-{attempt_id}.json", + f"evaluation-{attempt_id}.json", + f"review-attempt-{attempt_id}.json", + } + if names != expected_names: + raise ContractError("branch review handoff is missing required evidence artifacts") + _text(value["instructions"], "handoff instructions", maximum=2000) + return json.loads(canonical_json(value)) + + +def validate_handoff_decision_evidence( + decision: dict[str, Any], + packet: dict[str, Any], +) -> None: + """Require a lead decision to explicitly bind every presented evidence artifact.""" + if not isinstance(decision, dict) or not isinstance(decision.get("evidence_refs"), list): + raise ContractError("lead decision evidence_refs must be an array") + expected = { + (reference["artifact_id"], reference["sha256"]) + for reference in packet["artifacts"] + } + supplied = set() + for reference in decision["evidence_refs"]: + if (not isinstance(reference, dict) + or set(reference) != {"artifact_id", "sha256"} + or not isinstance(reference["artifact_id"], str) + or not isinstance(reference["sha256"], str)): + raise ContractError("lead decision evidence reference is invalid") + supplied.add((reference["artifact_id"], reference["sha256"])) + if len(decision["evidence_refs"]) != len(supplied) or supplied != expected: + raise ContractError("lead decision must bind every presented evidence artifact") diff --git a/plugin/core/src/devsquad/workspaces.py b/plugin/core/src/devsquad/workspaces.py index e062816..c94de21 100644 --- a/plugin/core/src/devsquad/workspaces.py +++ b/plugin/core/src/devsquad/workspaces.py @@ -148,7 +148,12 @@ def _validate_segment(value: str, label: str) -> str: def _validate_workspace( - source_repo: Path, workspace: Path, target_oid: str, scope_paths: Iterable[str], + source_repo: Path, + workspace: Path, + target_oid: str, + scope_paths: Iterable[str], + *, + require_clean: bool = True, ) -> None: try: resolved = workspace.resolve(strict=True) @@ -163,7 +168,7 @@ def _validate_workspace( raise ContractError("existing review workspace targets a different commit") if _git(resolved, "rev-parse", "--abbrev-ref", "HEAD").strip() != b"HEAD": raise ContractError("review workspace must use detached HEAD") - if dirty_paths(resolved): + if require_clean and dirty_paths(resolved): raise ContractError("existing review workspace is dirty") for scope in scope_paths: @@ -189,6 +194,27 @@ def _validate_workspace( raise ContractError(f"scoped symlink escapes the review workspace: {name}") +def reset_check_workspace( + review_workspace: Path, + check_workspace: Path, + target_oid: str, + scope_paths: Iterable[str], +) -> None: + """Reset only the exact run-owned check worktree before another check pass.""" + review = review_workspace.resolve(strict=True) + checks = check_workspace.resolve(strict=True) + if (review.name != "review-worktree" + or checks != review.parent / "check-worktree"): + raise ContractError("check workspace is not the review run's owned sibling") + _validate_workspace(review, review, target_oid, scope_paths) + _validate_workspace( + review, checks, target_oid, scope_paths, require_clean=False, + ) + _git(checks, "reset", "--hard", target_oid) + _git(checks, "clean", "-ffdx") + _validate_workspace(review, checks, target_oid, scope_paths) + + def _prepare_detached_workspace( source_repo: Path, runtime: Path, diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index bfa6bdd..a7e5e70 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -12,8 +12,9 @@ ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.contracts import ContractError from devsquad.service import Service -from devsquad.store import Store +from devsquad.store import ConflictError, Store, request_hash from devsquad_test_fixtures import branch_review_routing_documents @@ -103,6 +104,34 @@ def wait_state(self, run_id, states, timeout=10): detail = log.read_text() if log.exists() else "no supervisor log" self.fail(f"run did not reach {states}: {self.service.status(run_id)}\n{detail}") + @staticmethod + def decision(packet, submission_id, disposition, reason): + body = { + "schema_version": 1, + "submission_id": submission_id, + "disposition": disposition, + "reason": reason, + "evidence_refs": [ + { + "artifact_id": reference["artifact_id"], + "sha256": reference["sha256"], + } + for reference in packet["artifacts"] + ], + } + return {**body, "submission_hash": request_hash(body)} + + def start_waiting(self, key): + started = self.service.start( + self.task, + key, + _internal_review_fixture=self.fixture, + ) + self.assertTrue(started["created"]) + waiting = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + return started["run_id"], waiting + def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self): (self.repo / "notes.txt").write_text("unrelated local work\n") before_head = self.git_bytes("rev-parse", "HEAD") @@ -139,8 +168,12 @@ def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self) self.assertEqual([row["status"] for row in attempts], ["finished"]) artifacts = store.result_snapshot(started["run_id"])[1] names = {item["name"] for item in artifacts} + attempt_id = store.attempt(started["run_id"])["id"] self.assertTrue({ - "review.json", "checks.json", "evaluation.json", "review-attempt.json", + f"review-{attempt_id}.json", + f"checks-{attempt_id}.json", + f"evaluation-{attempt_id}.json", + f"review-attempt-{attempt_id}.json", } <= names) self.assertNotIn("result-receipt.json", names) handoff = store.handoff_snapshot(started["run_id"]) @@ -148,6 +181,7 @@ def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self) store.close() packet = handoff.packet + self.assertEqual(packet["attempt_id"], attempt_id) self.assertEqual(packet["candidate_sha256"], snapshot["workspace"]["candidate_sha256"]) self.assertEqual(packet["review"]["findings"][0]["id"], "F-1") self.assertEqual(packet["checks"][0]["status"], "failed") @@ -155,8 +189,8 @@ def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self) self.assertTrue(packet["evaluation"]["accept_allowed"]) self.assertEqual(packet["evaluation"]["report_only_failures"], ["fixture-tests"]) self.assertEqual( - {reference["name"] for reference in packet["artifacts"]}, - {"review.json", "checks.json", "evaluation.json", "review-attempt.json"}, + len(packet["artifacts"]), + 4, ) self.assertTrue(all(reference["artifact_id"] for reference in packet["artifacts"])) @@ -166,6 +200,180 @@ def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self) self.assertEqual(claimed["handoff"]["packet"], packet) self.assertFalse(self.service.result(started["run_id"])["ready"]) + decision = self.decision(packet, "accept-review", "accept", "Evidence accepted.") + completed = self.service.handoff_complete( + started["run_id"], claimed["claim"], decision, + ) + self.assertEqual((completed["state"], completed["phase"]), ("succeeded", None)) + self.assertEqual(completed["continuation"]["action"], "terminal") + self.assertFalse(completed["launched"]) + result = self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + names = {artifact["name"] for artifact in result["artifacts"]} + self.assertTrue({ + "receipt.json", "receipt.md", "events.jsonl", + "artifact-manifest.json", "result-receipt.json", + } <= names) + receipt_artifact = next( + artifact for artifact in result["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["state"], "succeeded") + self.assertEqual(receipt["candidate"]["sha256"], packet["candidate_sha256"]) + self.assertEqual(receipt["evaluation"]["report_only_failures"], ["fixture-tests"]) + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertIsNone(receipt["accounting"]["native_model_requests"]) + self.assertFalse(receipt["accounting"]["host_usage_measured"]) + self.assertEqual(receipt["lead"]["usage"]["source"], "unavailable") + self.assertFalse(receipt["events_export"]["includes_terminal_event"]) + self.assertIn("offline fixture", receipt["limitations"][0]) + replay = self.service.handoff_complete( + started["run_id"], claimed["claim"], decision, + ) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay["state"], "succeeded") + self.assertEqual( + (self.git_bytes("rev-parse", "HEAD"), + self.git_bytes("status", "--porcelain=v1", "-z"), + hashlib.sha256((self.repo / ".git/index").read_bytes()).hexdigest()), + (before_head, before_status, before_index), + ) + + def test_required_failure_blocks_accept_before_record_then_allows_reject(self): + self.task["checks"][0]["required_to_pass"] = True + run_id, waiting = self.start_waiting("required-failure") + claimed = self.service.handoff_claim(run_id, waiting["version"], "host-required") + packet = claimed["handoff"]["packet"] + accept = self.decision(packet, "blocked-accept", "accept", "Accept anyway.") + with self.assertRaisesRegex(ContractError, "blocked by required evidence"): + self.service.handoff_complete(run_id, claimed["claim"], accept) + status = self.service.status(run_id) + self.assertEqual((status["state"], status["phase"]), ("awaiting_host", None)) + + reject = self.decision(packet, "required-reject", "reject", "Required check failed.") + completed = self.service.handoff_complete(run_id, claimed["claim"], reject) + self.assertEqual(completed["state"], "failed") + result = self.service.result(run_id) + receipt_artifact = next( + artifact for artifact in result["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "REVIEW_REJECTED") + self.assertEqual(receipt["evaluation"]["required_failures"], ["fixture-tests"]) + + def test_decision_must_bind_every_presented_evidence_artifact(self): + run_id, waiting = self.start_waiting("missing-evidence") + claimed = self.service.handoff_claim(run_id, waiting["version"], "host-evidence") + packet = claimed["handoff"]["packet"] + decision = self.decision(packet, "missing-ref", "accept", "Incomplete evidence.") + decision["evidence_refs"].pop() + body = {key: value for key, value in decision.items() if key != "submission_hash"} + decision["submission_hash"] = request_hash(body) + with self.assertRaisesRegex(ContractError, "bind every presented"): + self.service.handoff_complete(run_id, claimed["claim"], decision) + self.assertEqual(self.service.status(run_id)["handoff"]["status"], "open") + + def test_revision_rechecks_clean_workspace_and_preserves_both_attempts(self): + self.task["budget"]["max_revisions"] = 1 + self.task["budget"]["max_worker_invocations"] = 3 + self.task["checks"] = [{ + "id": "dirtying-check", + "argv": [ + sys.executable, + "-c", + "from pathlib import Path; p=Path('generated.tmp'); " + "assert not p.exists(); p.write_text('generated')", + ], + "cwd": ".", + "timeout_seconds": 10, + "required_to_pass": True, + }] + run_id, first_wait = self.start_waiting("one-revision") + first_claim = self.service.handoff_claim( + run_id, first_wait["version"], "host-revision-one", + ) + first_packet = first_claim["handoff"]["packet"] + revise = self.decision( + first_packet, "request-revision", "revise", "Repeat the frozen review.", + ) + requeued = self.service.handoff_complete(run_id, first_claim["claim"], revise) + self.assertEqual(requeued["continuation"]["action"], "requeued") + self.assertTrue(requeued["launched"]) + + second_wait = self.wait_state(run_id, {"awaiting_host", "failed"}) + self.assertEqual(second_wait["state"], "awaiting_host") + self.assertEqual(second_wait["handoff"]["sequence"], 2) + second_claim = self.service.handoff_claim( + run_id, second_wait["version"], "host-revision-two", + ) + second_packet = second_claim["handoff"]["packet"] + self.assertNotEqual(first_packet["attempt_id"], second_packet["attempt_id"]) + self.assertTrue(all(result["status"] == "passed" for result in second_packet["checks"])) + self.assertTrue( + {reference["name"] for reference in first_packet["artifacts"]}.isdisjoint( + {reference["name"] for reference in second_packet["artifacts"]} + ) + ) + accept = self.decision( + second_packet, "accept-revision", "accept", "Second review accepted.", + ) + completed = self.service.handoff_complete(run_id, second_claim["claim"], accept) + self.assertEqual(completed["state"], "succeeded") + receipt_artifact = next( + artifact for artifact in self.service.result(run_id)["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], + ["revise", "accept"], + ) + self.assertEqual(receipt["revisions"]["executed"], 1) + late = self.decision( + first_packet, "late-first-host", "accept", "This claim is stale.", + ) + with self.assertRaises(ConflictError): + self.service.handoff_complete(run_id, first_claim["claim"], late) + + def test_zero_revision_budget_turns_revise_into_terminal_failure(self): + self.task["budget"]["max_revisions"] = 0 + run_id, waiting = self.start_waiting("no-revisions") + claimed = self.service.handoff_claim(run_id, waiting["version"], "host-no-revision") + packet = claimed["handoff"]["packet"] + revise = self.decision(packet, "revise-exhausted", "revise", "Try again.") + completed = self.service.handoff_complete(run_id, claimed["claim"], revise) + self.assertEqual(completed["state"], "failed") + receipt_artifact = next( + artifact for artifact in self.service.result(run_id)["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(receipt["lead"]["disposition"], "revise") + + def test_resume_finishes_submission_recorded_before_continuation(self): + run_id, waiting = self.start_waiting("resume-submission") + claimed = self.service.handoff_claim(run_id, waiting["version"], "host-crash") + packet = claimed["handoff"]["packet"] + decision = self.decision(packet, "saved-before-crash", "accept", "Accept evidence.") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + store.record_handoff_submission( + run_id, + self.service._decode_claim(claimed["claim"]), + decision, + ) + finally: + store.close() + self.assertEqual(self.service.status(run_id)["phase"], "handoff_submitted") + resumed = self.service.resume(run_id) + self.assertEqual((resumed["state"], resumed["disposition"]), ("succeeded", "terminal")) + self.assertFalse(resumed["launched"]) + self.assertTrue(self.service.result(run_id)["ready"]) + def test_invalid_internal_review_fails_before_launch_with_a_receipt(self): invalid = dict(self.fixture, verdict="clean") started = self.service.start( From 65a8d39b59f90ff436619198935f8734cae0cae5 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 17 Sep 2026 16:37:01 +0530 Subject: [PATCH 054/197] docs: checkpoint M3 host disposition pipeline --- docs/plans/engineering-team/RESUME.md | 48 +++++++++++++----------- docs/plans/engineering-team/backlog.json | 27 +++++++++++++ 2 files changed, 53 insertions(+), 22 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index abe0c69..f432c89 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,7 +2,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 16, 2026 +## Current position — September 17, 2026 - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` remains the published runtime @@ -22,19 +22,22 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne bounded target. `ddb6f51` fixes both: lease authorization now samples time after acquiring the SQLite write transaction, and public cancel resumes an interrupted `recovery_cleanup`. Both have deterministic regressions. -- M3 is in progress through `cd9a881`. `ca55990` adds strict deterministic +- M3 is in progress through `30cf49e`. `ca55990` adds strict deterministic profile/policy routing, alias binding snapshots, pins/fallbacks, independent reviewer selection and typed pool capacity. `1c7b614` freezes those exact bytes and decisions during public preflight. `cd9a881` resolves base/target OIDs and creates a detached run-owned review worktree without changing the - submitted checkout, index or HEAD; dirty scoped/config inputs and escaping - symlinks fail closed. The current gate is 144 core tests with - `ResourceWarning` promoted to failure plus 202 Bash assertions. + submitted checkout, index or HEAD. `a756307` defines strict review/check/ + evaluation evidence, `97d2c6c` runs the durable offline reviewer and trusted + checks, and `30cf49e` completes fenced host accept/revise/reject continuation, + bounded review retries and terminal JSON/Markdown/event/manifest reports. + The current gate is 164 core tests with `ResourceWarning` promoted to failure + plus 202 Bash assertions. - Public `branch-review` still fails explicitly as `CAPABILITY_UNAVAILABLE` - after the now-tested M3 preflight. The reviewer, declared checks, lead - disposition and bound report/receipt pipeline are not implemented yet. Do - not advertise DevSquad as ready for real review until that end-to-end gate - passes. + after the now-tested M3 preflight. The complete offline fixture path passes, + but a real provider reviewer adapter and bounded live Codex proof are still + required. Do not advertise DevSquad as ready for real review until that live + end-to-end gate passes. - Current provider readiness is external to M2: Claude CLI is not logged in; Grok CLI authentication expired; Gemini CLI's individual-account path is unsupported and its supported successor is Antigravity; Antigravity is @@ -60,7 +63,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 144 tests passed through the M3 frozen-workspace checkpoint | +| Python core discovery | 164 tests passed through M3 offline host disposition and reporting | | Bash 3.2 regression suite | 10 test files, 202 assertions passed | | Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 5 | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | @@ -68,6 +71,8 @@ Verified at the implementation/evidence checkpoints above: | M2 crash/race matrix | Real subprocess interruptions plus independent-process start, writer, cancel, import and host-handoff races passed at `ddb6f51` | | M3 deterministic routing | Strict profiles/policy, aliases, overrides, fallback and typed capacity tests passed at `ca55990` / `1c7b614` | | M3 frozen review input | Exact OIDs/config hashes, detached worktree, moving-ref stability, dirty-input rejection and source checkout preservation passed at `cd9a881` | +| M3 offline workflow evidence | Strict candidate-bound review/check evaluation and durable separate-worktree execution passed at `a756307` / `97d2c6c` | +| M3 host disposition/reporting | Accept/reject/revise, required-check blocking, retry budgets, stale claims, crash resume and five terminal reports passed at `30cf49e` | The first two saved-probe invocations failed before `Popen` because of probe-only path/field defects, so neither launched Codex nor consumed a model @@ -87,18 +92,17 @@ are not advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Preserve the M2 process/fencing boundaries and implement the next M3 slice: - the durable reviewer → trusted declared checks → lead disposition pipeline. - Add strict structured review/check/disposition artifacts bound to - `workspace.candidate_sha256`, explicit `required_to_pass` handling and host - handoff continuation. Do not bypass the supervisor with an in-process or - ad-hoc provider call. -3. Add terminal `receipt.json`, `receipt.md`, `events.jsonl` and artifact - manifest generation for the workflow. Keep public review unavailable until - the complete fake-adapter end-to-end gate passes, then run one bounded real - Codex review before declaring the M3 product stop usable. -4. Use offline fake adapters for development. Provider login/permission work is - a later live gate and must not block independent M3 implementation. +2. Preserve the M2 process/fencing boundaries and add the real Codex reviewer + launch through the verified native adapter. Feed it the frozen prompt and + workspace, preserve requested/observed identity and strict JSON evidence, + and keep review-only permissions. Do not bypass the supervisor with an + in-process or ad-hoc provider call. +3. Add focused adapter fault tests for denied writes, missing/malformed output, + protocol interruption and usage/accounting. Then run one bounded real Codex + branch review and save a redacted receipt before declaring the M3 product + stop usable. +4. Keep Claude/Grok/Antigravity probes paused until their normal login or trust + blockers are resolved. They do not block the independent Codex M3 gate. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index a2d8f9d..796095e 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -111,6 +111,33 @@ "artifact": "../../../test/core/test_workspaces.py", "recorded_at": "2026-09-16T01:08:00+05:30", "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "a756307", + "command_or_action": "154 core tests with ResourceWarning promoted to error and 202 shell assertions covering strict candidate-bound review, check, evaluation and attempt evidence", + "outcome": "Reviewer output and trusted check results are schema-bound to the frozen candidate; required failures cannot be overridden and usage remains explicitly unknown when unavailable", + "artifact": "../../../test/core/test_review_workflow.py", + "recorded_at": "2026-09-16T01:21:23+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "97d2c6c", + "command_or_action": "159 core tests with ResourceWarning promoted to error and 202 shell assertions covering durable offline review, isolated declared checks and atomic host handoff publication", + "outcome": "A frozen offline reviewer and separate check worktree now produce durable evidence and a claimable host packet without changing the submitted checkout", + "artifact": "../../../test/core/test_review_runtime.py", + "recorded_at": "2026-09-16T01:34:41+05:30", + "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "30cf49e", + "command_or_action": "164 core tests with ResourceWarning promoted to error and 202 shell assertions covering host accept/reject/revise, retry budgets, exact evidence binding, crash resume and terminal reports", + "outcome": "The offline M3 path now completes end to end with attempt-specific evidence, required-check enforcement, clean retry workspaces and durable receipt.json, receipt.md, events.jsonl, artifact manifest and result receipt; real provider execution remains pending", + "artifact": "../../../test/core/test_review_runtime.py", + "recorded_at": "2026-09-17T16:35:24+05:30", + "availability": "tracked_tests" } ], "blocker": null From 459ff3fc7783d3a2603daa454232939e037346fe Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 17 Sep 2026 16:55:21 +0530 Subject: [PATCH 055/197] feat: run M3 reviews through native Codex --- plugin/core/pyproject.toml | 2 +- plugin/core/src/devsquad/codex_protocol.py | 26 +- .../core/src/devsquad/codex_review_worker.py | 420 ++++++++++++++++++ plugin/core/src/devsquad/detached.py | 19 +- plugin/core/src/devsquad/review_worker.py | 39 +- plugin/core/src/devsquad/service.py | 31 +- plugin/core/src/devsquad/supervisor.py | 4 +- plugin/core/src/devsquad/workflows.py | 93 +++- test/core/fakes/codex_review_cli.py | 156 +++++++ test/core/test_m1.py | 24 + test/core/test_review_runtime.py | 142 ++++++ 11 files changed, 913 insertions(+), 43 deletions(-) create mode 100644 plugin/core/src/devsquad/codex_review_worker.py create mode 100755 test/core/fakes/codex_review_cli.py diff --git a/plugin/core/pyproject.toml b/plugin/core/pyproject.toml index 63b4eff..f9bf72e 100644 --- a/plugin/core/pyproject.toml +++ b/plugin/core/pyproject.toml @@ -25,5 +25,5 @@ where = ["src"] "share/devsquad/adapters/antigravity" = ["adapters/antigravity/adapter.json"] "share/devsquad/adapters/grok" = ["adapters/grok/adapter.json"] "share/devsquad/adapters" = ["adapters/classification-policy.conf"] -"share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/profiles.schema.json", "schemas/task.schema.json"] +"share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/check-result.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/profiles.schema.json", "schemas/review-result.schema.json", "schemas/task.schema.json"] "share/devsquad/profiles" = ["profiles/templates.json"] diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index 49552a6..4db9c5e 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -70,11 +70,20 @@ def initialized_notification() -> dict[str, Any]: return {"method": "initialized", "params": {}} -def thread_start_request(request_id: int, *, cwd: str, model: str, permission: str) -> dict[str, Any]: +def thread_start_request( + request_id: int, + *, + cwd: str, + model: str, + permission: str, + ephemeral: bool = False, +) -> dict[str, Any]: sandbox = {"read_only": "read-only", "workspace_write": "workspace-write"}.get(permission) if sandbox is None: raise ContractError(f"unsupported native permission: {permission}") - return request(request_id, "thread/start", {"cwd": cwd, "model": model, "sandbox": sandbox, "approvalPolicy": "never", "ephemeral": False}) + if type(ephemeral) is not bool: + raise ContractError("native thread ephemeral flag must be boolean") + return request(request_id, "thread/start", {"cwd": cwd, "model": model, "sandbox": sandbox, "approvalPolicy": "never", "ephemeral": ephemeral}) def model_list_request(request_id: int, cursor: str | None = None, limit: int = 100) -> dict[str, Any]: @@ -126,7 +135,7 @@ def consume(self, message: dict[str, Any]) -> None: self.events.append(message) method = message.get("method", "") params = message.get("params", message.get("result", {})) - if method in {"thread/started", "thread/start/completed", "turn/started", "item/agentMessage/delta", "turn/output/delta", "turn/completed", "error"} and not isinstance(params, dict): + if method in {"thread/started", "thread/start/completed", "turn/started", "item/agentMessage/delta", "turn/output/delta", "item/completed", "turn/completed", "error"} and not isinstance(params, dict): raise ContractError("native event params must be an object") if not isinstance(params, dict): return @@ -162,6 +171,17 @@ def consume(self, message: dict[str, Any]) -> None: raise ContractError("native output delta must be a string") if self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: self.output.append(delta) + if (method == "item/completed" and not self.output + and self.thread_id and self.turn_id + and message_thread == self.thread_id and message_turn == self.turn_id): + item = params.get("item") + if not isinstance(item, dict): + raise ContractError("native completed item must be an object") + if item.get("type") in {"agentMessage", "agent_message"}: + content = item.get("text", item.get("content")) + if not isinstance(content, str): + raise ContractError("native completed agent message must contain text") + self.output.append(content) if method == "turn/completed" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: self.terminal = True self.terminal_status = turn.get("status") diff --git a/plugin/core/src/devsquad/codex_review_worker.py b/plugin/core/src/devsquad/codex_review_worker.py new file mode 100644 index 0000000..ef664f2 --- /dev/null +++ b/plugin/core/src/devsquad/codex_review_worker.py @@ -0,0 +1,420 @@ +"""Drive one frozen read-only Codex app-server review inside the M2 worker.""" + +from __future__ import annotations + +import hashlib +import io +import json +import os +from pathlib import Path +import subprocess +import sys +import time +from typing import Any + +from .catalog import normalize_models, verified_efforts +from .codex_protocol import ( + JsonLinePeer, + NativeTurnState, + discover_models, + initialize_request, + initialized_notification, + receive_response, + thread_start_request, + turn_start_request, +) +from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported +from .review_worker import run_review_and_checks +from .store import canonical_json +from .supervisor import BoundedDrain +from .workflows import ( + MAX_REVIEW_BYTES, + build_review_prompt, + decode_review_document, + review_output_schema, +) + + +MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 +MAX_PROTOCOL_EVENT_BYTES = 4 * 1024 * 1024 +MAX_PROTOCOL_EVENTS = 10_000 +MAX_SERVER_STDERR_BYTES = 256 * 1024 +ADAPTER_FIELDS = { + "schema_version", "harness", "transport", "binary", "binary_sha256", + "harness_version", "model_provider", "output_schema_sha256", +} + + +def _adapter_manifest_path() -> Path: + source = Path(__file__).resolve().parents[2] / "adapters" / "codex" / "adapter.json" + if source.is_file(): + return source + installed = Path(sys.prefix) / "share" / "devsquad" / "adapters" / "codex" / "adapter.json" + if installed.is_file(): + return installed + raise CapabilityUnavailable("Codex adapter manifest is unavailable") + + +def freeze_codex_reviewer(selected: dict[str, Any]) -> dict[str, Any]: + """Resolve and verify the non-model Codex launch identity during preflight.""" + from .adapters import AdapterManifest, harness_version + + if not isinstance(selected, dict) or not isinstance(selected.get("profile"), dict): + raise ContractError("frozen reviewer selection is invalid") + profile = selected["profile"] + if profile.get("harness") != "codex": + raise CapabilityUnavailable( + f"selected reviewer harness is not implemented for M3: {profile.get('harness')}" + ) + if profile.get("permission_policy") != "read_only": + raise ProfileUnsupported("Codex branch review requires read_only permission") + if set(profile.get("required_tools", [])) - {"read"}: + raise ProfileUnsupported("Codex branch review profile requests unsupported tools") + effort = profile.get("effort") + if (not isinstance(effort, dict) or effort.get("transport") != "native" + or not isinstance(effort.get("value"), str) or not effort["value"]): + raise ProfileUnsupported("Codex branch review requires an explicit native effort") + if not isinstance(profile.get("model_id"), str) or not profile["model_id"]: + raise ProfileUnsupported("Codex branch review requires an exact model id") + + manifest = AdapterManifest.load(_adapter_manifest_path()) + if manifest.name != "codex" or manifest.transport != "native_protocol": + raise CapabilityUnavailable("installed Codex adapter is not native_protocol") + binary_name = manifest.resolve_binary() + if not binary_name: + raise CapabilityUnavailable("Codex executable is unavailable") + binary = Path(binary_name).resolve(strict=True) + version = harness_version(str(binary)) + if not version: + raise CapabilityUnavailable("Codex version could not be observed") + if version not in manifest.verified_versions: + raise ProfileUnsupported(f"unverified Codex app-server version: {version}") + content = binary.read_bytes() + schema_hash = hashlib.sha256(canonical_json(review_output_schema()).encode()).hexdigest() + return { + "schema_version": 1, + "harness": "codex", + "transport": "native_protocol", + "binary": str(binary), + "binary_sha256": hashlib.sha256(content).hexdigest(), + "harness_version": version, + "model_provider": manifest.model_provider or "openai", + "output_schema_sha256": schema_hash, + } + + +def _validated_adapter(snapshot: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any]]: + adapter = snapshot.get("review_adapter") + try: + selected = snapshot["routing"]["roles"]["reviewer"]["selected"] + profile = selected["profile"] + except (KeyError, TypeError) as exc: + raise ContractError("frozen Codex reviewer selection is missing") from exc + if not isinstance(adapter, dict) or set(adapter) != ADAPTER_FIELDS: + raise ContractError("frozen Codex review adapter fields are invalid") + if (adapter["schema_version"] != 1 or type(adapter["schema_version"]) is not int + or adapter["harness"] != "codex" + or adapter["transport"] != "native_protocol" + or adapter["model_provider"] != "openai"): + raise ContractError("frozen Codex review adapter identity is invalid") + for field in ("binary", "binary_sha256", "harness_version", "output_schema_sha256"): + if not isinstance(adapter[field], str) or not adapter[field]: + raise ContractError("frozen Codex review adapter value is invalid") + if (len(adapter["binary_sha256"]) != 64 + or len(adapter["output_schema_sha256"]) != 64): + raise ContractError("frozen Codex review adapter hash is invalid") + if (not isinstance(profile, dict) or profile.get("harness") != "codex" + or profile.get("permission_policy") != "read_only"): + raise ContractError("frozen profile is not a read-only Codex reviewer") + schema_hash = hashlib.sha256(canonical_json(review_output_schema()).encode()).hexdigest() + if schema_hash != adapter["output_schema_sha256"]: + raise ContractError("frozen review output schema changed") + return adapter, profile + + +def _remaining(deadline: float) -> float: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("native Codex review exceeded its deadline") + return remaining + + +def _response_result(response: dict[str, Any], label: str) -> dict[str, Any]: + if "error" in response: + raise ContractError(f"{label} failed: {response['error']}") + result = response.get("result") + if not isinstance(result, dict): + raise ContractError(f"{label} returned no result") + return result + + +def _usage(events: list[dict[str, Any]], thread_id: str, turn_id: str) -> dict[str, Any]: + for event in reversed(events): + if event.get("method") != "thread/tokenUsage/updated": + continue + params = event.get("params") + if (not isinstance(params, dict) or params.get("threadId") != thread_id + or params.get("turnId") not in {None, turn_id}): + continue + token_usage = params.get("tokenUsage") + total = token_usage.get("total") if isinstance(token_usage, dict) else None + if not isinstance(total, dict): + continue + values = [total.get(key) for key in ("inputTokens", "outputTokens", "totalTokens")] + if all(type(value) is int and value >= 0 for value in values): + return { + "input_tokens": values[0], + "output_tokens": values[1], + "total_tokens": values[2], + "source": "native_reported", + } + return { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + } + + +def _stop_server( + process: subprocess.Popen[bytes], + writer: io.TextIOWrapper | None, + stderr: BoundedDrain | None, +) -> None: + if writer is not None: + try: + writer.close() + except OSError: + pass + if process.poll() is None: + process.terminate() + try: + process.wait(timeout=2) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=2) + if stderr is not None: + metadata = stderr.finish() + if stderr.content: + sys.stderr.buffer.write(bytes(stderr.content)) + if metadata["truncated"]: + sys.stderr.write( + f"\n[Codex stderr truncated; total_bytes={metadata['total_bytes']} " + f"sha256={metadata['full_sha256']}]\n" + ) + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("workflow snapshot must be an object") + adapter, profile = _validated_adapter(snapshot) + binary = Path(adapter["binary"]) + try: + resolved = binary.resolve(strict=True) + except OSError as exc: + raise CapabilityUnavailable("frozen Codex executable is missing") from exc + if (resolved != binary + or hashlib.sha256(binary.read_bytes()).hexdigest() != adapter["binary_sha256"]): + raise CapabilityUnavailable("frozen Codex executable changed after preflight") + try: + version = subprocess.run( + [str(binary), "--version"], + text=True, + capture_output=True, + timeout=3, + check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise CapabilityUnavailable( + "frozen Codex version could not be re-observed" + ) from exc + if version.returncode != 0 or version.stdout.strip() != adapter["harness_version"]: + raise CapabilityUnavailable("Codex version changed after preflight") + + effort = profile["effort"]["value"] + model = profile["model_id"] + workspace = snapshot["workspace"] + review_root = Path(workspace["path"]).resolve(strict=True) + argv = [ + str(binary), + "-c", f'model="{model}"', + "-c", f'model_reasoning_effort="{effort}"', + "app-server", "--listen", "stdio://", + ] + process: subprocess.Popen[bytes] | None = None + writer = None + stderr = None + protocol_events: list[dict[str, Any]] = [] + protocol_bytes = 0 + deadline = time.monotonic() + snapshot["task"]["budget"]["wall_seconds"] + + def record(message: dict[str, Any]) -> None: + nonlocal protocol_bytes + encoded = canonical_json(message).encode() + protocol_bytes += len(encoded) + if (len(protocol_events) >= MAX_PROTOCOL_EVENTS + or protocol_bytes > MAX_PROTOCOL_EVENT_BYTES): + raise ContractError("native Codex protocol evidence exceeds its bound") + protocol_events.append(message) + + try: + process = subprocess.Popen( + argv, + cwd=review_root, + env=os.environ.copy(), + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + start_new_session=False, + close_fds=True, + ) + assert process.stdin is not None and process.stdout is not None and process.stderr is not None + writer = io.TextIOWrapper(process.stdin, encoding="utf-8", write_through=True) + peer = JsonLinePeer(process.stdout, writer) + stderr = BoundedDrain(process.stderr, MAX_SERVER_STDERR_BYTES) + stderr.start() + + peer.send(initialize_request(1)) + _response_result( + receive_response( + peer, 1, timeout_seconds=_remaining(deadline), + on_notification=record, + ), + "Codex initialize", + ) + peer.send(initialized_notification()) + models = discover_models( + peer, + first_request_id=10, + timeout_seconds=_remaining(deadline), + ) + catalog = { + "complete": True, + "harness": "codex", + "harness_version": adapter["harness_version"], + "models": normalize_models("codex", adapter["harness_version"], models), + } + if effort not in verified_efforts( + catalog, + harness="codex", + version=adapter["harness_version"], + model_id=model, + ): + raise ProfileUnsupported( + f"unsupported or unverified effort {effort!r} for codex model {model!r}" + ) + + peer.send(thread_start_request( + 100, + cwd=str(review_root), + model=model, + permission="read_only", + ephemeral=True, + )) + thread_result = _response_result( + receive_response( + peer, 100, timeout_seconds=_remaining(deadline), + on_notification=record, + ), + "Codex thread/start", + ) + thread = thread_result.get("thread") + thread_id = thread.get("id") if isinstance(thread, dict) else None + reported_version = thread.get("cliVersion") if isinstance(thread, dict) else None + if not isinstance(thread_id, str) or not thread_id: + raise ContractError("Codex thread/start returned no thread id") + if reported_version and f"codex-cli {reported_version}" != adapter["harness_version"]: + raise ContractError("Codex thread reported a different harness version") + expected_policy = {"type": "readOnly", "networkAccess": False} + reported_cwd = thread_result.get("cwd") + if not isinstance(reported_cwd, str) or not reported_cwd: + raise ContractError("Codex thread returned an invalid working directory") + if (thread_result.get("model") != model + or thread_result.get("reasoningEffort") != effort + or thread_result.get("modelProvider") != adapter["model_provider"] + or thread_result.get("approvalPolicy") != "never" + or thread_result.get("sandbox") != expected_policy + or Path(reported_cwd).resolve() != review_root): + raise ContractError("Codex thread did not preserve the frozen execution identity") + + prompt = build_review_prompt(snapshot["task"], workspace) + peer.send(turn_start_request( + 101, + thread_id=thread_id, + prompt=prompt, + model=model, + effort=effort, + cwd=str(review_root), + permission="read_only", + output_schema=review_output_schema(), + )) + turn_result = _response_result( + receive_response( + peer, 101, timeout_seconds=_remaining(deadline), + on_notification=record, + ), + "Codex turn/start", + ) + turn = turn_result.get("turn") + turn_id = turn.get("id") if isinstance(turn, dict) else None + if not isinstance(turn_id, str) or not turn_id: + raise ContractError("Codex turn/start returned no turn id") + state = NativeTurnState(thread_id=thread_id, turn_id=turn_id) + for event in protocol_events: + state.consume(event) + output_bytes = sum(len(part.encode()) for part in state.output) + while not state.terminal: + message = peer.receive(_remaining(deadline)) + if "id" in message and "method" in message: + raise ContractError("native Codex requested an unsupported host action") + record(message) + prior = len(state.output) + state.consume(message) + output_bytes += sum(len(part.encode()) for part in state.output[prior:]) + if output_bytes > MAX_REVIEW_BYTES: + raise ContractError("native Codex review output exceeds its byte limit") + if state.terminal_status != "completed": + raise ContractError( + f"native Codex review did not complete successfully: {state.terminal_status}" + ) + review = decode_review_document( + "".join(state.output).strip(), snapshot["task"], workspace, + ) + usage = _usage(protocol_events, thread_id, turn_id) + finally: + if process is not None: + _stop_server(process, writer, stderr) + + observed_identity = { + "harness": "codex", + "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], + "model_id": model, + "effort": effort, + "permission_policy": "read_only", + "verification": "verified", + } + return run_review_and_checks( + snapshot, + review, + observed_identity=observed_identity, + native_ids={"thread_id": thread_id, "turn_id": turn_id}, + native_model_requests=None, + usage=usage, + ) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_SNAPSHOT_BYTES + 1) + if len(payload) > MAX_SNAPSHOT_BYTES: + raise ContractError("workflow snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("workflow snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index ceece57..d2d729f 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -28,21 +28,26 @@ def main(argv=None): "DEVSQUAD_DELEGATION_DEPTH": "1", } stdin_path = None - if "internal_review_fixture" in snapshot: + if "internal_review_fixture" in snapshot or "review_adapter" in snapshot: selected = snapshot["routing"]["roles"]["reviewer"]["selected"]["profile"] + adapter = snapshot.get("review_adapter") identity = ExecutionIdentity( - "devsquad-review-workflow", - "1", - None, + selected["harness"], + adapter["harness_version"] if adapter else "fixture", + adapter["model_provider"] if adapter else None, selected["model_family"], selected["model_id"], selected["effort"]["value"], tuple(selected["required_tools"]), selected["permission_policy"], selected["account_pool_id"], - "unknown", + "verified" if adapter else "unknown", ) - command = [sys.executable, "-P", "-m", "devsquad.review_worker"] + module = ( + "devsquad.codex_review_worker" + if adapter else "devsquad.review_worker" + ) + command = [sys.executable, "-P", "-m", module] input_path, _, _ = store.finalize_artifact( args.run_id, "workflow-input.json", @@ -57,7 +62,7 @@ def main(argv=None): spec = LaunchSpec( 1, identity.harness, - "cli_exec", + "native_protocol" if "review_adapter" in snapshot else "cli_exec", tuple(command), run["worktree_path"], stdin_path, diff --git a/plugin/core/src/devsquad/review_worker.py b/plugin/core/src/devsquad/review_worker.py index eed3e34..f6cf36f 100644 --- a/plugin/core/src/devsquad/review_worker.py +++ b/plugin/core/src/devsquad/review_worker.py @@ -14,7 +14,11 @@ from .contracts import ContractError from .store import canonical_json from .supervisor import BoundedDrain -from .workflows import MAX_PREVIEW_CHARS, make_branch_review_evidence +from .workflows import ( + MAX_PREVIEW_CHARS, + make_branch_review_evidence, + validate_review_document, +) from .workspaces import dirty_paths, reset_check_workspace @@ -133,17 +137,18 @@ def _verify_finding_locations(review: dict[str, Any], workspace: Path) -> None: ) -def run(snapshot: dict[str, Any]) -> dict[str, Any]: - if not isinstance(snapshot, dict): - raise ContractError("workflow snapshot must be an object") +def run_review_and_checks( + snapshot: dict[str, Any], + review: dict[str, Any], + **attempt_metadata: Any, +) -> dict[str, Any]: + """Validate one review, run the frozen checks, and build combined evidence.""" task = snapshot.get("task") workspace = snapshot.get("workspace") check_workspace = snapshot.get("check_workspace") - fixture = snapshot.get("internal_review_fixture") - if not all(isinstance(value, dict) for value in ( - task, workspace, check_workspace, fixture, - )): - raise ContractError("offline review snapshot is incomplete") + if not all(isinstance(value, dict) for value in (task, workspace, check_workspace)): + raise ContractError("branch review snapshot is incomplete") + normalized_review = validate_review_document(review, task, workspace) review_root = Path(workspace["path"]).resolve(strict=True) checks_root = Path(check_workspace["path"]).resolve(strict=True) reset_check_workspace( @@ -154,8 +159,7 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: ) if dirty_paths(review_root): raise ContractError("frozen review workspace is dirty before reviewer execution") - review = fixture - _verify_finding_locations(review, review_root) + _verify_finding_locations(normalized_review, review_root) if dirty_paths(review_root): raise ContractError("reviewer modified the frozen read-only workspace") checks = [ @@ -167,7 +171,18 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: ) for check in task["checks"] ] - return make_branch_review_evidence(snapshot, review, checks) + return make_branch_review_evidence( + snapshot, normalized_review, checks, **attempt_metadata, + ) + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("workflow snapshot must be an object") + fixture = snapshot.get("internal_review_fixture") + if not isinstance(fixture, dict): + raise ContractError("offline review snapshot is incomplete") + return run_review_and_checks(snapshot, fixture) def main() -> int: diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index b14e152..1a755b0 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -13,7 +13,8 @@ import threading from typing import Any -from .contracts import ContractError +from .codex_review_worker import freeze_codex_reviewer +from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported from .reports import build_terminal_reports from .router import load_routing from .store import ( @@ -281,18 +282,9 @@ def _continue_preparation( internal_review_fixture=internal_review_fixture, ) if internal_delay is None and internal_review_fixture is None: - error = { - "error": "CAPABILITY_UNAVAILABLE", - "message": "branch-review workflow is introduced in M3", - } - store.fail_preparation( - run_id, - fencing_token, - error, - mutable_snapshot=snapshot, - supersedes_run_id=validated_supersedes_run_id, + snapshot["review_adapter"] = freeze_codex_reviewer( + snapshot["routing"]["roles"]["reviewer"]["selected"], ) - return None, error package, digest = self._freeze_package() version = store.complete_preparation( run_id, @@ -301,12 +293,19 @@ def _continue_preparation( package_path=str(package), package_digest=digest, supersedes_run_id=supersedes_run_id, - worktree_path=( - snapshot["workspace"]["path"] - if internal_review_fixture is not None else None - ), + worktree_path=(snapshot.get("workspace") or {}).get("path"), ) return (version, package, digest), None + except (CapabilityUnavailable, ProfileUnsupported) as exc: + error = {"error": exc.code, "message": str(exc)} + store.fail_preparation( + run_id, + fencing_token, + error, + mutable_snapshot=snapshot, + supersedes_run_id=validated_supersedes_run_id, + ) + return None, error except Exception as exc: error = {"error": "PREPARATION_FAILED", "message": str(exc)} try: diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index e3b0ac7..1db6566 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -403,7 +403,9 @@ def import_durable(self, run_id: str) -> str: path,digest,size=self.store.finalize_artifact(run_id,logical,data) artifacts.append({"name":logical,"path":path,"sha256":digest,"byte_size":size}) snapshot=json.loads(self.store.run(run_id)["mutable_snapshot"]) - workflow_review="internal_review_fixture" in snapshot + workflow_review = ( + "internal_review_fixture" in snapshot or "review_adapter" in snapshot + ) semantic_error=None if (workflow_review and not receipt["cancelled"] and not receipt["timed_out"] and receipt["returncode"]==0): diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index 549bca5..feb85fc 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -25,6 +25,55 @@ _COMMIT_OID = re.compile(r"[0-9a-f]{40}\Z") +def review_output_schema() -> dict[str, Any]: + """Return the strict native structured-output schema for one review.""" + return { + "type": "object", + "additionalProperties": False, + "required": [ + "schema_version", "candidate_sha256", "base_oid", "target_oid", + "review_mode", "verdict", "summary", "findings", + ], + "properties": { + "schema_version": {"const": 1}, + "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "base_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "target_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, + "review_mode": {"enum": ["standard", "adversarial"]}, + "verdict": {"enum": ["clean", "findings"]}, + "summary": {"type": "string", "minLength": 1, "maxLength": 20_000}, + "findings": { + "type": "array", + "maxItems": MAX_FINDINGS, + "items": { + "type": "object", + "additionalProperties": False, + "required": [ + "id", "severity", "title", "description", "path", + "start_line", "end_line", "evidence", + ], + "properties": { + "id": {"type": "string", "minLength": 1, "maxLength": 200}, + "severity": {"enum": sorted(FINDING_SEVERITIES)}, + "title": {"type": "string", "minLength": 1, "maxLength": 500}, + "description": { + "type": "string", "minLength": 1, + "maxLength": MAX_TEXT_CHARS, + }, + "path": {"type": "string", "minLength": 1}, + "start_line": {"type": "integer", "minimum": 1}, + "end_line": {"type": "integer", "minimum": 1}, + "evidence": { + "type": "string", "minLength": 1, + "maxLength": MAX_TEXT_CHARS, + }, + }, + }, + }, + }, + } + + def _exact( value: Any, fields: set[str], @@ -434,7 +483,8 @@ def validate_branch_review_evidence( raise ContractError("branch review evaluation does not match derived gates") attempt = _exact(document["attempt"], { "role", "selected_profile", "prompt_sha256", "review_sha256", - "worker_invocations", "native_model_requests", "usage", + "observed_identity", "native_ids", "worker_invocations", + "native_model_requests", "usage", }, "review attempt evidence") if attempt["role"] != "reviewer": raise ContractError("review attempt role is invalid") @@ -450,6 +500,36 @@ def validate_branch_review_evidence( review_sha256 = hashlib.sha256(canonical_json(review).encode()).hexdigest() if _sha256(attempt["review_sha256"], "review document sha256") != review_sha256: raise ContractError("review document hash is invalid") + adapter = snapshot.get("review_adapter") + observed = attempt["observed_identity"] + native_ids = attempt["native_ids"] + if adapter is None: + if observed is not None or native_ids != {}: + raise ContractError("fixture review cannot claim a native observed identity") + else: + if not isinstance(adapter, dict): + raise ContractError("frozen review adapter is invalid") + for field in ("harness", "harness_version", "model_provider"): + if not isinstance(adapter.get(field), str) or not adapter[field]: + raise ContractError("frozen review adapter identity is invalid") + identity = _exact(observed, { + "harness", "harness_version", "model_provider", "model_id", "effort", + "permission_policy", "verification", + }, "observed reviewer identity") + expected_identity = { + "harness": adapter["harness"], + "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], + "model_id": frozen_reviewer["profile"]["model_id"], + "effort": frozen_reviewer["profile"]["effort"]["value"], + "permission_policy": frozen_reviewer["profile"]["permission_policy"], + "verification": "verified", + } + if canonical_json(identity) != canonical_json(expected_identity): + raise ContractError("observed reviewer identity does not match the frozen adapter") + ids = _exact(native_ids, {"thread_id", "turn_id"}, "native reviewer ids") + for field in ("thread_id", "turn_id"): + _text(ids[field], f"native reviewer {field}", maximum=500) if attempt["worker_invocations"] != 1 or type(attempt["worker_invocations"]) is not int: raise ContractError("review worker invocation accounting is invalid") if attempt["worker_invocations"] > task["budget"]["max_worker_invocations"]: @@ -479,6 +559,11 @@ def make_branch_review_evidence( snapshot: dict[str, Any], review: dict[str, Any], checks: list[dict[str, Any]], + *, + observed_identity: dict[str, Any] | None = None, + native_ids: dict[str, str] | None = None, + native_model_requests: int | None = None, + usage: dict[str, Any] | None = None, ) -> dict[str, Any]: task, workspace = snapshot["task"], snapshot["workspace"] normalized_review = validate_review_document(review, task, workspace) @@ -504,9 +589,11 @@ def make_branch_review_evidence( "review_sha256": hashlib.sha256( canonical_json(normalized_review).encode() ).hexdigest(), + "observed_identity": observed_identity, + "native_ids": native_ids if native_ids is not None else {}, "worker_invocations": 1, - "native_model_requests": None, - "usage": { + "native_model_requests": native_model_requests, + "usage": usage if usage is not None else { "input_tokens": None, "output_tokens": None, "total_tokens": None, diff --git a/test/core/fakes/codex_review_cli.py b/test/core/fakes/codex_review_cli.py new file mode 100755 index 0000000..07cc8f5 --- /dev/null +++ b/test/core/fakes/codex_review_cli.py @@ -0,0 +1,156 @@ +#!/usr/bin/env python3 +"""Small Codex app-server fixture used by the public M3 native review tests.""" + +import json +import re +import sys + + +if "--version" in sys.argv: + print("codex-cli 0.135.0") + raise SystemExit(0) + + +model = "gpt-fake-review" +for index, argument in enumerate(sys.argv[:-1]): + if argument == "-c": + match = re.fullmatch(r'model="([^"]+)"', sys.argv[index + 1]) + if match: + model = match.group(1) + + +initialized = False +thread_id = "fixture-thread" +turn_id = "fixture-turn" +for line in sys.stdin: + request = json.loads(line) + method = request.get("method") + request_id = request.get("id") + params = request.get("params", {}) + if method == "initialize": + print(json.dumps({ + "id": request_id, + "result": {"serverInfo": {"name": "fake-codex", "version": "0.135.0"}}, + }), flush=True) + elif method == "initialized": + initialized = True + elif method == "model/list": + if not initialized: + print(json.dumps({ + "id": request_id, + "error": {"code": -32002, "message": "not initialized"}, + }), flush=True) + continue + print(json.dumps({ + "id": request_id, + "result": { + "data": [{ + "id": model, + "supportedReasoningEfforts": [{ + "reasoningEffort": "low", + "description": "fixture", + }], + }], + "nextCursor": None, + }, + }), flush=True) + elif method == "thread/start": + sandbox = ( + {"type": "workspaceWrite", "writableRoots": [params["cwd"]], "networkAccess": False} + if model.endswith("identity-drift") + else {"type": "readOnly", "networkAccess": False} + ) + print(json.dumps({ + "id": request_id, + "result": { + "thread": { + "id": thread_id, + "cliVersion": "0.135.0", + "modelProvider": "openai", + }, + "model": params["model"], + "modelProvider": "openai", + "reasoningEffort": "low", + "cwd": params["cwd"], + "sandbox": sandbox, + "approvalPolicy": "never", + "activePermissionProfile": None, + }, + }), flush=True) + elif method == "turn/start": + if model.endswith("denied"): + print(json.dumps({ + "id": request_id, + "error": {"code": -32000, "message": "permission denied by fixture"}, + }), flush=True) + continue + prompt = params["input"][0]["text"] + assignment = json.loads(prompt.splitlines()[-1]) + review = { + "schema_version": 1, + "candidate_sha256": assignment["candidate_sha256"], + "base_oid": assignment["base_oid"], + "target_oid": assignment["target_oid"], + "review_mode": assignment["review_mode"], + "verdict": "clean", + "summary": "The bounded native fixture found no supported defect.", + "findings": [], + } + output = ( + "{}" if model.endswith("malformed") + else json.dumps(review, sort_keys=True, separators=(",", ":")) + ) + print(json.dumps({ + "id": request_id, + "result": {"turn": {"id": turn_id}}, + }), flush=True) + if model.endswith("disconnect"): + break + print(json.dumps({ + "method": "turn/started", + "params": { + "threadId": thread_id, + "turn": {"id": turn_id, "status": "inProgress"}, + }, + }), flush=True) + midpoint = len(output) // 2 + for part in (output[:midpoint], output[midpoint:]): + print(json.dumps({ + "method": "item/agentMessage/delta", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "delta": part, + }, + }), flush=True) + print(json.dumps({ + "method": "thread/tokenUsage/updated", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "tokenUsage": { + "total": { + "inputTokens": 120, + "cachedInputTokens": 0, + "outputTokens": 40, + "reasoningOutputTokens": 10, + "totalTokens": 160, + }, + "last": { + "inputTokens": 120, + "cachedInputTokens": 0, + "outputTokens": 40, + "reasoningOutputTokens": 10, + "totalTokens": 160, + }, + "modelContextWindow": 1000, + }, + }, + }), flush=True) + print(json.dumps({ + "method": "turn/completed", + "params": { + "threadId": thread_id, + "turn": {"id": turn_id, "status": "completed"}, + }, + }), flush=True) diff --git a/test/core/test_m1.py b/test/core/test_m1.py index eb5fbcf..6e2a392 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -190,6 +190,25 @@ def test_start_and_interrupt_ack_are_not_terminal(self): state.consume({"method": "turn/completed", "params": {"threadId":"th1", "turn":{"id":"t1", "status":"interrupted"}}}) self.assertTrue(state.terminal) + def test_completed_agent_message_is_output_fallback_without_duplication(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + state.consume({ + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"type": "agentMessage", "text": "fallback"}, + }, + }) + self.assertEqual(state.output, ["fallback"]) + state.consume({ + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"type": "agentMessage", "text": "duplicate"}, + }, + }) + self.assertEqual(state.output, ["fallback"]) + def test_unrelated_turn_cannot_complete_ours_and_disconnect_is_visible(self): state = NativeTurnState(thread_id="th1", turn_id="ours") state.consume({"method":"turn/completed", "params":{"threadId":"th1", "turn":{"id":"other", "status":"completed"}}}) @@ -199,6 +218,11 @@ def test_unrelated_turn_cannot_complete_ours_and_disconnect_is_visible(self): def test_typed_native_requests_match_installed_contract(self): thread = thread_start_request(1, cwd="/tmp/repo", model="gpt-test", permission="read_only") self.assertEqual(thread["params"]["sandbox"], "read-only") + self.assertFalse(thread["params"]["ephemeral"]) + self.assertTrue(thread_start_request( + 5, cwd="/tmp/repo", model="gpt-test", permission="read_only", + ephemeral=True, + )["params"]["ephemeral"]) turn = turn_start_request(2, thread_id="th", prompt="p", model="gpt-test", effort="low", cwd="/tmp/repo", permission="read_only", output_schema={"type":"object"}) self.assertEqual(turn["params"]["sandboxPolicy"], {"type":"readOnly", "networkAccess":False}) self.assertEqual(turn["params"]["outputSchema"]["type"], "object") diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index a7e5e70..199e7a1 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -2,12 +2,14 @@ import hashlib import json +import os from pathlib import Path import subprocess import sys import tempfile import time import unittest +from unittest.mock import patch ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "plugin/core/src")) @@ -374,6 +376,146 @@ def test_resume_finishes_submission_recorded_before_continuation(self): self.assertFalse(resumed["launched"]) self.assertTrue(self.service.result(run_id)["ready"]) + def test_public_native_codex_driver_verifies_identity_usage_and_output(self): + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "native-codex-reviewer", + "harness": "codex", + "model_family": "gpt-fixture", + "model_id": "gpt-fake-review", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "codex-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["native-fixture"], + }], + "bindings": { + "review.deep": {"profile_id": "native-codex-reviewer", "version": 1}, + }, + } + policy_document = { + "schema_version": 1, + "id": "native-codex-policy", + "version": 1, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"fixture-review-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "codex-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy_document, sort_keys=True) + "\n" + ) + subprocess.run( + ["git", "-C", str(self.repo), "add", "devsquad"], check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "native review config"], + check=True, + ) + self.target = self.git_text("rev-parse", "HEAD").strip() + self.task["project"]["target_ref"] = self.target + fake_bin = self.root / "fake-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to( + ROOT / "test/core/fakes/codex_review_cli.py" + ) + environment = { + "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + } + with patch.dict(os.environ, environment, clear=False): + started = self.service.start(self.task, "native-codex-review") + self.assertTrue(started["created"]) + waiting = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + claimed = self.service.handoff_claim( + started["run_id"], waiting["version"], "native-host", + ) + packet = claimed["handoff"]["packet"] + observed = packet["attempt"]["observed_identity"] + self.assertEqual( + (observed["harness"], observed["harness_version"], observed["model_id"]), + ("codex", "codex-cli 0.135.0", "gpt-fake-review"), + ) + self.assertEqual(packet["attempt"]["usage"], { + "input_tokens": 120, + "output_tokens": 40, + "total_tokens": 160, + "source": "native_reported", + }) + self.assertEqual(packet["review"]["verdict"], "clean") + decision = self.decision(packet, "accept-native", "accept", "Native review accepted.") + completed = self.service.handoff_complete( + started["run_id"], claimed["claim"], decision, + ) + self.assertEqual(completed["state"], "succeeded") + receipt_artifact = next( + artifact for artifact in self.service.result(started["run_id"])["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["accounting"]["attempt_usage"][0]["total_tokens"], 160) + self.assertEqual(receipt["limitations"], []) + + def test_native_codex_faults_never_become_valid_reviews(self): + profiles_text, policy_text = branch_review_routing_documents() + profiles = json.loads(profiles_text) + profile = profiles["profiles"][0] + profile.update({ + "harness": "codex", + "model_family": "gpt-fixture", + "required_tools": ["read"], + "permission_policy": "read_only", + }) + fake_bin = self.root / "fault-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to( + ROOT / "test/core/fakes/codex_review_cli.py" + ) + environment = { + "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + } + for mode in ("malformed", "denied", "disconnect", "identity-drift"): + with self.subTest(mode=mode): + profile["model_id"] = f"gpt-fake-{mode}" + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text(policy_text) + subprocess.run( + ["git", "-C", str(self.repo), "add", "devsquad"], check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", f"fault {mode}"], + check=True, + ) + target = self.git_text("rev-parse", "HEAD").strip() + self.task["project"]["target_ref"] = target + with patch.dict(os.environ, environment, clear=False): + started = self.service.start(self.task, f"native-fault-{mode}") + self.assertEqual(started["state"], "queued") + failed = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(failed["state"], "failed") + self.assertIsNone(failed["handoff"]) + result = self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + self.assertNotIn( + "receipt.json", {artifact["name"] for artifact in result["artifacts"]}, + ) + def test_invalid_internal_review_fails_before_launch_with_a_receipt(self): invalid = dict(self.fixture, verdict="clean") started = self.service.start( From 3cfb409831a16146b2476961acd1b5ddc03f1475 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 17 Sep 2026 16:57:16 +0530 Subject: [PATCH 056/197] docs: checkpoint native Codex review path --- docs/plans/engineering-team/RESUME.md | 41 +++++++++++++----------- docs/plans/engineering-team/backlog.json | 9 ++++++ 2 files changed, 32 insertions(+), 18 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index f432c89..7dfb3a8 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -22,7 +22,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne bounded target. `ddb6f51` fixes both: lease authorization now samples time after acquiring the SQLite write transaction, and public cancel resumes an interrupted `recovery_cleanup`. Both have deterministic regressions. -- M3 is in progress through `30cf49e`. `ca55990` adds strict deterministic +- M3 is in progress through `459ff3f`. `ca55990` adds strict deterministic profile/policy routing, alias binding snapshots, pins/fallbacks, independent reviewer selection and typed pool capacity. `1c7b614` freezes those exact bytes and decisions during public preflight. `cd9a881` resolves base/target @@ -31,13 +31,18 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne evaluation evidence, `97d2c6c` runs the durable offline reviewer and trusted checks, and `30cf49e` completes fenced host accept/revise/reject continuation, bounded review retries and terminal JSON/Markdown/event/manifest reports. - The current gate is 164 core tests with `ResourceWarning` promoted to failure - plus 202 Bash assertions. -- Public `branch-review` still fails explicitly as `CAPABILITY_UNAVAILABLE` - after the now-tested M3 preflight. The complete offline fixture path passes, - but a real provider reviewer adapter and bounded live Codex proof are still - required. Do not advertise DevSquad as ready for real review until that live - end-to-end gate passes. + `459ff3f` adds the public native Codex reviewer: it freezes the verified CLI, + launches one ephemeral read-only app-server turn inside the existing M2 + process group, verifies observed model/effort/sandbox identity, accepts only + strict candidate-bound JSON, and records native usage. The current gate is + 167 core tests with `ResourceWarning` promoted to failure plus 202 Bash + assertions. +- Public `branch-review` now reaches the complete native Codex path in offline + protocol tests. Malformed output, denial, disconnect and identity drift all + fail without becoming valid reviews. A bounded real Codex end-to-end proof + is still required before advertising it as ready for real review. M3 also + still needs rich reports for early terminal failures, materialized waiting + handoff reports and the configured headless-lead path. - Current provider readiness is external to M2: Claude CLI is not logged in; Grok CLI authentication expired; Gemini CLI's individual-account path is unsupported and its supported successor is Antigravity; Antigravity is @@ -63,7 +68,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 164 tests passed through M3 offline host disposition and reporting | +| Python core discovery | 167 tests passed through the M3 native Codex reviewer path | | Bash 3.2 regression suite | 10 test files, 202 assertions passed | | Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 5 | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | @@ -73,6 +78,7 @@ Verified at the implementation/evidence checkpoints above: | M3 frozen review input | Exact OIDs/config hashes, detached worktree, moving-ref stability, dirty-input rejection and source checkout preservation passed at `cd9a881` | | M3 offline workflow evidence | Strict candidate-bound review/check evaluation and durable separate-worktree execution passed at `a756307` / `97d2c6c` | | M3 host disposition/reporting | Accept/reject/revise, required-check blocking, retry budgets, stale claims, crash resume and five terminal reports passed at `30cf49e` | +| M3 native Codex reviewer | Public start, exact identity verification, ephemeral read-only structured output, native usage and four provider-fault classes passed offline at `459ff3f` | The first two saved-probe invocations failed before `Popen` because of probe-only path/field defects, so neither launched Codex nor consumed a model @@ -92,15 +98,14 @@ are not advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Preserve the M2 process/fencing boundaries and add the real Codex reviewer - launch through the verified native adapter. Feed it the frozen prompt and - workspace, preserve requested/observed identity and strict JSON evidence, - and keep review-only permissions. Do not bypass the supervisor with an - in-process or ad-hoc provider call. -3. Add focused adapter fault tests for denied writes, missing/malformed output, - protocol interruption and usage/accounting. Then run one bounded real Codex - branch review and save a redacted receipt before declaring the M3 product - stop usable. +2. Add a reproducible, opt-in bounded live M3 probe that uses the public + service, keeps raw provider material under `~/.devsquad/private-probes`, and + emits only a redacted receipt/hash. Run it once with the existing + subscription-backed Codex login and retain the evidence. +3. Close the remaining M3 contract gaps: rich terminal reports for failures + before host disposition, materialized waiting handoff reports, and the + configured headless-lead path. Rerun the full core/Bash gates and audit M3 + against its acceptance section before marking it complete. 4. Keep Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. They do not block the independent Codex M3 gate. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 796095e..7c8f6db 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -138,6 +138,15 @@ "artifact": "../../../test/core/test_review_runtime.py", "recorded_at": "2026-09-17T16:35:24+05:30", "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "459ff3f", + "command_or_action": "167 core tests with ResourceWarning promoted to error and 202 shell assertions covering the public native Codex reviewer and provider protocol faults", + "outcome": "M3 can run one frozen, ephemeral, read-only Codex app-server review inside the M2 process fence, verify exact execution identity, validate candidate-bound structured output and record native usage; the bounded real Codex gate remains pending", + "artifact": "../../../test/core/test_review_runtime.py", + "recorded_at": "2026-09-17T16:55:37+05:30", + "availability": "tracked_tests" } ], "blocker": null From aaf91b9ccee034a522307240fb66bf984635aab7 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 17 Sep 2026 17:01:02 +0530 Subject: [PATCH 057/197] test: add bounded M3 Codex live probe --- test/core/probes/m3_codex_review.py | 345 ++++++++++++++++++++++++++++ 1 file changed, 345 insertions(+) create mode 100755 test/core/probes/m3_codex_review.py diff --git a/test/core/probes/m3_codex_review.py b/test/core/probes/m3_codex_review.py new file mode 100755 index 0000000..d0b2efe --- /dev/null +++ b/test/core/probes/m3_codex_review.py @@ -0,0 +1,345 @@ +#!/usr/bin/env python3 +"""Opt-in live proof for the public M3 native Codex branch-review path.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import subprocess +import sys +import time +from typing import Any + + +ROOT = Path(__file__).resolve().parents[3] +CORE_SRC = ROOT / "plugin" / "core" / "src" +sys.path.insert(0, str(CORE_SRC)) + +from devsquad.service import Service +from devsquad.store import request_hash + + +TERMINAL_STATES = {"succeeded", "failed", "cancelled", "timed_out"} +SAFE_MODEL = re.compile(r"[A-Za-z0-9._-]+\Z") + + +def _git(repo: Path, *arguments: str) -> str: + return subprocess.run( + ["git", "-C", str(repo), *arguments], + text=True, + capture_output=True, + check=True, + ).stdout.strip() + + +def _write_json(path: Path, value: dict[str, Any]) -> None: + path.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(65536), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _wait(service: Service, run_id: str, timeout: int) -> dict[str, Any]: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] in TERMINAL_STATES | {"awaiting_host", "blocked"}: + return status + time.sleep(0.1) + raise TimeoutError("public M3 review did not reach a bounded handoff or terminal state") + + +def _make_repository(repo: Path, model: str, effort: str) -> tuple[str, str]: + repo.mkdir() + _git(repo, "init", "-q") + _git(repo, "config", "user.email", "devsquad-probe@example.invalid") + _git(repo, "config", "user.name", "DevSquad Probe") + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "devsquad").mkdir() + (repo / "src/__init__.py").write_text("") + (repo / "src/ratio.py").write_text( + "def safe_ratio(numerator, denominator):\n" + " if denominator == 0:\n" + " return None\n" + " return numerator / denominator\n" + ) + (repo / "tests/test_ratio.py").write_text( + "import unittest\n" + "from src.ratio import safe_ratio\n\n" + "class RatioTest(unittest.TestCase):\n" + " def test_positive_ratio(self):\n" + " self.assertEqual(safe_ratio(6, 3), 2)\n" + ) + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "m3-live-codex-reviewer", + "harness": "codex", + "model_family": "gpt", + "model_id": model, + "effort": {"value": effort, "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "codex-subscription", + "billing_mode": "subscription", + "quality_status": "trial", + "evidence_refs": ["m3-live-probe"], + }], + "bindings": { + "review.deep": { + "profile_id": "m3-live-codex-reviewer", + "version": 1, + }, + }, + } + policy = { + "schema_version": 1, + "id": "m3-live-probe-policy", + "version": 1, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"live-review-small": "trial"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "codex-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + _write_json(repo / "devsquad/profiles.json", profiles) + _write_json(repo / "devsquad/policy.json", policy) + _git(repo, "add", ".") + _git(repo, "commit", "-qm", "probe base") + base = _git(repo, "rev-parse", "HEAD") + + # Deliberately introduce a small reviewable regression while leaving the + # declared positive-path check green. The integration proof does not + # require a particular verdict, but a supported finding is useful evidence. + (repo / "src/ratio.py").write_text( + "def safe_ratio(numerator, denominator):\n" + " if numerator == 0:\n" + " return None\n" + " return numerator / denominator\n" + ) + _git(repo, "add", "src/ratio.py") + _git(repo, "commit", "-qm", "candidate regression") + return base, _git(repo, "rev-parse", "HEAD") + + +def _task(repo: Path, base: str, target: str, wall_seconds: int) -> dict[str, Any]: + return { + "schema_version": 1, + "project": { + "repo_path": str(repo), + "base_ref": base, + "target_ref": target, + }, + "workflow": "branch-review", + "goal": "Review the exact candidate for correctness regressions with file-and-line evidence.", + "task_class": "live-review-small", + "acceptance": [{ + "id": "candidate-bound-review", + "description": "Return a candidate-bound review with supported file locations.", + "evidence_kind": "review", + }, { + "id": "declared-check", + "description": "Record the trusted declared check outcome.", + "evidence_kind": "check", + }], + "checks": [{ + "id": "unit-tests", + "argv": [sys.executable, "-m", "unittest", "discover", "-s", "tests", "-q"], + "cwd": ".", + "timeout_seconds": 30, + "required_to_pass": True, + }], + "scope": {"read_paths": ["src", "tests"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": wall_seconds, + "max_worker_invocations": 1, + "max_revisions": 0, + "max_fallbacks_per_step": 0, + }, + "review": {"mode": "standard"}, + "origin": {"surface": "live-probe"}, + } + + +def _decision(packet: dict[str, Any]) -> dict[str, Any]: + body = { + "schema_version": 1, + "submission_id": "m3-live-host-accept", + "disposition": "accept", + "reason": "Accept the bounded reviewer and required-check evidence.", + "evidence_refs": [{ + "artifact_id": reference["artifact_id"], + "sha256": reference["sha256"], + } for reference in packet["artifacts"]], + } + return {**body, "submission_hash": request_hash(body)} + + +def _lock_down(root: Path) -> None: + for path in sorted(root.rglob("*"), reverse=True): + try: + if path.is_dir(): + path.chmod(0o700) + elif path.is_file(): + path.chmod(0o600) + except OSError: + pass + root.chmod(0o700) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--run-live", action="store_true", + help="required acknowledgement for one subscription-backed model turn", + ) + parser.add_argument( + "--output-dir", type=Path, + default=Path.home() / ".devsquad" / "private-probes", + ) + parser.add_argument("--model", default="gpt-5.5") + parser.add_argument( + "--effort", default="low", + choices=("minimal", "low", "medium", "high", "xhigh"), + ) + parser.add_argument("--timeout", type=int, default=180) + args = parser.parse_args() + if not args.run_live: + parser.error("--run-live is required") + if not SAFE_MODEL.fullmatch(args.model): + parser.error("--model must contain only letters, numbers, dot, underscore or hyphen") + if args.timeout < 30 or args.timeout > 600: + parser.error("--timeout must be between 30 and 600 seconds") + + stamp = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + revision = _git(ROOT, "rev-parse", "HEAD") + run_dir = ( + args.output_dir.expanduser().resolve() + / f"m3-codex-review-{stamp}-{revision[:12]}" + ) + run_dir.mkdir(parents=True, mode=0o700, exist_ok=False) + run_dir.chmod(0o700) + receipt: dict[str, Any] = { + "schema_version": 1, + "status": "failed", + "started_at": stamp, + "revision": revision, + "requested": {"model": args.model, "effort": args.effort}, + } + service: Service | None = None + run_id: str | None = None + try: + repo = run_dir / "repository" + runtime = run_dir / "runtime" + base, target = _make_repository(repo, args.model, args.effort) + service = Service(runtime) + started = service.start( + _task(repo, base, target, args.timeout), + f"m3-live-{stamp}", + ) + run_id = started["run_id"] + receipt["run_id"] = run_id + if started["state"] == "failed": + raise RuntimeError(f"public start failed: {started.get('error')}") + waiting = _wait(service, run_id, args.timeout + 30) + if waiting["state"] != "awaiting_host": + raise RuntimeError(f"review did not produce a host handoff: {waiting}") + claimed = service.handoff_claim( + run_id, waiting["version"], "m3-live-probe-host", + ) + packet = claimed["handoff"]["packet"] + attempt = packet["attempt"] + observed = attempt["observed_identity"] + if (observed["harness"] != "codex" + or observed["model_id"] != args.model + or observed["effort"] != args.effort + or observed["permission_policy"] != "read_only" + or observed["verification"] != "verified"): + raise RuntimeError(f"observed reviewer identity drifted: {observed}") + if attempt["usage"]["source"] != "native_reported": + raise RuntimeError("Codex did not report native token usage") + if packet["evaluation"]["accept_allowed"] is not True: + raise RuntimeError(f"required evidence blocked acceptance: {packet['evaluation']}") + completed = service.handoff_complete( + run_id, claimed["claim"], _decision(packet), + ) + if completed["state"] != "succeeded": + raise RuntimeError(f"host completion did not succeed: {completed}") + result = service.result(run_id) + artifacts = {item["name"]: item for item in result["artifacts"]} + required_reports = { + "receipt.json", "receipt.md", "events.jsonl", + "artifact-manifest.json", "result-receipt.json", + } + if not result["ready"] or not required_reports <= set(artifacts): + raise RuntimeError("terminal result is missing required M3 reports") + receipt.update({ + "status": "passed", + "observed": observed, + "native_ids_present": { + key: bool(attempt["native_ids"].get(key)) + for key in ("thread_id", "turn_id") + }, + "usage": attempt["usage"], + "review": { + "verdict": packet["review"]["verdict"], + "finding_count": len(packet["review"]["findings"]), + }, + "checks": [{ + "id": check["id"], "status": check["status"], + "required_to_pass": check["required_to_pass"], + } for check in packet["checks"]], + "candidate_sha256": packet["candidate_sha256"], + "terminal_state": completed["state"], + "report_sha256": { + name: artifacts[name]["sha256"] for name in sorted(required_reports) + }, + }) + except Exception as exc: + receipt["error_type"] = type(exc).__name__ + receipt["error"] = str(exc) + finally: + if service is not None and run_id is not None: + try: + status = service.status(run_id) + if status["state"] not in TERMINAL_STATES: + receipt["cleanup"] = service.cancel(run_id) + receipt["cleanup_terminal"] = _wait(service, run_id, 15)["state"] + except Exception as cleanup_error: + receipt["cleanup_error"] = str(cleanup_error) + receipt["finished_at"] = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + receipt_path = run_dir / "probe-receipt.json" + _write_json(receipt_path, receipt) + _lock_down(run_dir) + print(json.dumps({ + "status": receipt["status"], + "run_dir": str(run_dir), + "receipt_sha256": _sha256(receipt_path), + }, sort_keys=True)) + return 0 if receipt["status"] == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) From e10db947d04a607d0d3ae831a7527abf302f015b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 17 Sep 2026 17:11:00 +0530 Subject: [PATCH 058/197] fix: isolate and refresh native Codex reviews --- plugin/core/adapters/codex/adapter.json | 2 +- plugin/core/src/devsquad/codex_protocol.py | 5 ++ .../core/src/devsquad/codex_review_worker.py | 68 +++++++++++++++++-- test/core/fakes/codex_review_cli.py | 6 +- test/core/probes/m3_codex_review.py | 13 ++++ test/core/test_m1.py | 23 +++++-- test/core/test_review_runtime.py | 12 +++- 7 files changed, 116 insertions(+), 13 deletions(-) diff --git a/plugin/core/adapters/codex/adapter.json b/plugin/core/adapters/codex/adapter.json index 9a1d757..67fb311 100644 --- a/plugin/core/adapters/codex/adapter.json +++ b/plugin/core/adapters/codex/adapter.json @@ -5,7 +5,7 @@ "fallback_transport": "cli_exec", "binary_candidates": ["codex"], "model_provider": "openai", - "verified_harness_versions": ["codex-cli 0.135.0"], + "verified_harness_versions": ["codex-cli 0.153.4"], "capabilities": {"efforts_by_model": {}, "native_model_list": true, "resume": true}, "permission_profiles": {"read_only": [], "workspace_write": []}, "output_format": "jsonl" diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index 4db9c5e..4372744 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -185,6 +185,11 @@ def consume(self, message: dict[str, Any]) -> None: if method == "turn/completed" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: self.terminal = True self.terminal_status = turn.get("status") + turn_error = turn.get("error") + if turn_error is not None: + if not isinstance(turn_error, dict): + raise ContractError("native terminal turn error must be an object") + self.error = turn_error if method == "error" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id and not params.get("willRetry", False): self.terminal = True self.terminal_status = "failed" diff --git a/plugin/core/src/devsquad/codex_review_worker.py b/plugin/core/src/devsquad/codex_review_worker.py index ef664f2..7536511 100644 --- a/plugin/core/src/devsquad/codex_review_worker.py +++ b/plugin/core/src/devsquad/codex_review_worker.py @@ -9,6 +9,7 @@ from pathlib import Path import subprocess import sys +import tempfile import time from typing import Any @@ -41,7 +42,7 @@ MAX_SERVER_STDERR_BYTES = 256 * 1024 ADAPTER_FIELDS = { "schema_version", "harness", "transport", "binary", "binary_sha256", - "harness_version", "model_provider", "output_schema_sha256", + "harness_version", "model_provider", "output_schema_sha256", "auth_file", } @@ -55,6 +56,26 @@ def _adapter_manifest_path() -> Path: raise CapabilityUnavailable("Codex adapter manifest is unavailable") +def _subscription_auth_file(value: str | None = None) -> Path: + try: + if value is None: + configured_home = os.environ.get("CODEX_HOME") + root = Path(configured_home).expanduser() if configured_home else Path.home() / ".codex" + auth_file = (root / "auth.json").resolve(strict=True) + else: + auth_file = Path(value).resolve(strict=True) + metadata = auth_file.stat() + except OSError as exc: + raise CapabilityUnavailable("Codex subscription auth file is unavailable") from exc + if not auth_file.is_file(): + raise CapabilityUnavailable("Codex subscription auth path is not a file") + if hasattr(os, "getuid") and metadata.st_uid != os.getuid(): + raise CapabilityUnavailable("Codex subscription auth file is not user-owned") + if os.name != "nt" and metadata.st_mode & 0o077: + raise CapabilityUnavailable("Codex subscription auth file permissions are too broad") + return auth_file + + def freeze_codex_reviewer(selected: dict[str, Any]) -> dict[str, Any]: """Resolve and verify the non-model Codex launch identity during preflight.""" from .adapters import AdapterManifest, harness_version @@ -100,6 +121,7 @@ def freeze_codex_reviewer(selected: dict[str, Any]) -> dict[str, Any]: "harness_version": version, "model_provider": manifest.model_provider or "openai", "output_schema_sha256": schema_hash, + "auth_file": str(_subscription_auth_file()), } @@ -117,12 +139,17 @@ def _validated_adapter(snapshot: dict[str, Any]) -> tuple[dict[str, Any], dict[s or adapter["transport"] != "native_protocol" or adapter["model_provider"] != "openai"): raise ContractError("frozen Codex review adapter identity is invalid") - for field in ("binary", "binary_sha256", "harness_version", "output_schema_sha256"): + for field in ( + "binary", "binary_sha256", "harness_version", "output_schema_sha256", + "auth_file", + ): if not isinstance(adapter[field], str) or not adapter[field]: raise ContractError("frozen Codex review adapter value is invalid") if (len(adapter["binary_sha256"]) != 64 or len(adapter["output_schema_sha256"]) != 64): raise ContractError("frozen Codex review adapter hash is invalid") + if not Path(adapter["auth_file"]).is_absolute(): + raise ContractError("frozen Codex auth path is not absolute") if (not isinstance(profile, dict) or profile.get("harness") != "codex" or profile.get("permission_policy") != "read_only"): raise ContractError("frozen profile is not a read-only Codex reviewer") @@ -204,6 +231,24 @@ def _stop_server( ) +def _isolated_codex_environment( + auth_file_value: str, +) -> tuple[tempfile.TemporaryDirectory, dict[str, str]]: + """Expose subscription auth without loading user sessions, config or plugins.""" + auth_file = _subscription_auth_file(auth_file_value) + home = tempfile.TemporaryDirectory(prefix="devsquad-codex-home-") + root = Path(home.name) + try: + root.chmod(0o700) + (root / "auth.json").symlink_to(auth_file) + except Exception: + home.cleanup() + raise + environment = os.environ.copy() + environment["CODEX_HOME"] = str(root) + return home, environment + + def run(snapshot: dict[str, Any]) -> dict[str, Any]: if not isinstance(snapshot, dict): raise ContractError("workflow snapshot must be an object") @@ -239,11 +284,18 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: str(binary), "-c", f'model="{model}"', "-c", f'model_reasoning_effort="{effort}"', + "--disable", "apps", + "--disable", "plugins", + "--disable", "browser_use", + "--disable", "computer_use", + "--disable", "multi_agent", + "--enable", "skip_host_skill_discovery", "app-server", "--listen", "stdio://", ] process: subprocess.Popen[bytes] | None = None writer = None stderr = None + codex_home: tempfile.TemporaryDirectory | None = None protocol_events: list[dict[str, Any]] = [] protocol_bytes = 0 deadline = time.monotonic() + snapshot["task"]["budget"]["wall_seconds"] @@ -258,10 +310,11 @@ def record(message: dict[str, Any]) -> None: protocol_events.append(message) try: + codex_home, environment = _isolated_codex_environment(adapter["auth_file"]) process = subprocess.Popen( argv, cwd=review_root, - env=os.environ.copy(), + env=environment, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, @@ -374,8 +427,13 @@ def record(message: dict[str, Any]) -> None: if output_bytes > MAX_REVIEW_BYTES: raise ContractError("native Codex review output exceeds its byte limit") if state.terminal_status != "completed": + detail = ( + canonical_json(state.error)[:2000] + if state.error is not None else "no provider error was reported" + ) raise ContractError( - f"native Codex review did not complete successfully: {state.terminal_status}" + "native Codex review did not complete successfully: " + f"{state.terminal_status}; {detail}" ) review = decode_review_document( "".join(state.output).strip(), snapshot["task"], workspace, @@ -384,6 +442,8 @@ def record(message: dict[str, Any]) -> None: finally: if process is not None: _stop_server(process, writer, stderr) + if codex_home is not None: + codex_home.cleanup() observed_identity = { "harness": "codex", diff --git a/test/core/fakes/codex_review_cli.py b/test/core/fakes/codex_review_cli.py index 07cc8f5..6d75d41 100755 --- a/test/core/fakes/codex_review_cli.py +++ b/test/core/fakes/codex_review_cli.py @@ -7,7 +7,7 @@ if "--version" in sys.argv: - print("codex-cli 0.135.0") + print("codex-cli 0.153.4") raise SystemExit(0) @@ -30,7 +30,7 @@ if method == "initialize": print(json.dumps({ "id": request_id, - "result": {"serverInfo": {"name": "fake-codex", "version": "0.135.0"}}, + "result": {"serverInfo": {"name": "fake-codex", "version": "0.153.4"}}, }), flush=True) elif method == "initialized": initialized = True @@ -65,7 +65,7 @@ "result": { "thread": { "id": thread_id, - "cliVersion": "0.135.0", + "cliVersion": "0.153.4", "modelProvider": "openai", }, "model": params["model"], diff --git a/test/core/probes/m3_codex_review.py b/test/core/probes/m3_codex_review.py index d0b2efe..80e84d3 100755 --- a/test/core/probes/m3_codex_review.py +++ b/test/core/probes/m3_codex_review.py @@ -220,6 +220,10 @@ def main() -> int: default=Path.home() / ".devsquad" / "private-probes", ) parser.add_argument("--model", default="gpt-5.5") + parser.add_argument( + "--codex-binary", type=Path, + help="optional absolute codex binary; its directory is prepended for this probe only", + ) parser.add_argument( "--effort", default="low", choices=("minimal", "low", "medium", "high", "xhigh"), @@ -232,6 +236,15 @@ def main() -> int: parser.error("--model must contain only letters, numbers, dot, underscore or hyphen") if args.timeout < 30 or args.timeout > 600: parser.error("--timeout must be between 30 and 600 seconds") + if args.codex_binary is not None: + try: + binary = args.codex_binary.expanduser().resolve(strict=True) + except OSError as exc: + parser.error(f"--codex-binary cannot be resolved: {exc}") + if (not binary.is_file() or binary.name != "codex" + or not os.access(binary, os.X_OK)): + parser.error("--codex-binary must be an executable absolute path named codex") + os.environ["PATH"] = f"{binary.parent}{os.pathsep}{os.environ.get('PATH', '')}" stamp = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) revision = _git(ROOT, "rev-parse", "HEAD") diff --git a/test/core/test_m1.py b/test/core/test_m1.py index 6e2a392..070d4c2 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -94,7 +94,7 @@ def test_native_launch_is_preparation_only_and_version_scoped(self): temp, binary = self.fake_path("codex"); self.addCleanup(temp.cleanup) manifest = self.manifest("codex").with_model_efforts({"gpt-test": ("low",)}) with patch.dict(os.environ, {"PATH": str(binary.parent)}): - spec = prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.135.0") + spec = prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.153.4") self.assertEqual(spec.transport, "native_protocol") self.assertEqual(spec.argv[-2:], ("--listen", "stdio://")) self.assertIn('model_reasoning_effort="low"', spec.argv) @@ -105,13 +105,13 @@ def test_native_launch_is_preparation_only_and_version_scoped(self): def test_discovered_snapshot_feeds_native_preparation_and_rejects_drift(self): temp, binary = self.fake_path("codex"); self.addCleanup(temp.cleanup) manifest = self.manifest("codex") - snapshot = {"complete":True,"harness":"codex","harness_version":"codex-cli 0.135.0","models":[{"id":"gpt-test","supported_efforts":["low"]}]} + snapshot = {"complete":True,"harness":"codex","harness_version":"codex-cli 0.153.4","models":[{"id":"gpt-test","supported_efforts":["low"]}]} with patch.dict(os.environ, {"PATH": str(binary.parent)}): - spec = prepare_native_codex_from_catalog(manifest, snapshot, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.135.0") + spec = prepare_native_codex_from_catalog(manifest, snapshot, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.153.4") self.assertEqual(spec.requested.model, "gpt-test") drifted = dict(snapshot); drifted["harness_version"] = "codex-cli future" with self.assertRaises(ContractError): - prepare_native_codex_from_catalog(manifest, drifted, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.135.0") + prepare_native_codex_from_catalog(manifest, drifted, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.153.4") def test_launch_round_trip_is_strict(self): temp, binary = self.fake_path("grok"); self.addCleanup(temp.cleanup) @@ -215,6 +215,21 @@ def test_unrelated_turn_cannot_complete_ours_and_disconnect_is_visible(self): self.assertFalse(state.terminal) state.disconnected(); self.assertEqual(state.terminal_status, "transport_disconnected") + def test_failed_terminal_turn_preserves_typed_provider_error(self): + state = NativeTurnState(thread_id="th1", turn_id="ours") + state.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": { + "id": "ours", "status": "failed", + "error": {"code": "model_error", "message": "bounded detail"}, + }, + }, + }) + self.assertEqual(state.terminal_status, "failed") + self.assertEqual(state.error["code"], "model_error") + def test_typed_native_requests_match_installed_contract(self): thread = thread_start_request(1, cwd="/tmp/repo", model="gpt-test", permission="read_only") self.assertEqual(thread["params"]["sandbox"], "read-only") diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index 199e7a1..6b13a71 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -433,8 +433,13 @@ def test_public_native_codex_driver_verifies_identity_usage_and_output(self): (fake_bin / "codex").symlink_to( ROOT / "test/core/fakes/codex_review_cli.py" ) + fake_home = self.root / "fake-codex-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) environment = { "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + "CODEX_HOME": str(fake_home), } with patch.dict(os.environ, environment, clear=False): started = self.service.start(self.task, "native-codex-review") @@ -448,7 +453,7 @@ def test_public_native_codex_driver_verifies_identity_usage_and_output(self): observed = packet["attempt"]["observed_identity"] self.assertEqual( (observed["harness"], observed["harness_version"], observed["model_id"]), - ("codex", "codex-cli 0.135.0", "gpt-fake-review"), + ("codex", "codex-cli 0.153.4", "gpt-fake-review"), ) self.assertEqual(packet["attempt"]["usage"], { "input_tokens": 120, @@ -485,8 +490,13 @@ def test_native_codex_faults_never_become_valid_reviews(self): (fake_bin / "codex").symlink_to( ROOT / "test/core/fakes/codex_review_cli.py" ) + fake_home = self.root / "fault-codex-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) environment = { "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + "CODEX_HOME": str(fake_home), } for mode in ("malformed", "denied", "disconnect", "identity-drift"): with self.subTest(mode=mode): From 9478796247486ead4e62578cff4035f418534ac7 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 17 Sep 2026 17:14:12 +0530 Subject: [PATCH 059/197] fix: type native review output schema --- plugin/core/src/devsquad/workflows.py | 28 +++++++++++++-------------- test/core/test_review_workflow.py | 17 ++++++++++++++++ 2 files changed, 30 insertions(+), 15 deletions(-) diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index feb85fc..9b3edc0 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -35,13 +35,15 @@ def review_output_schema() -> dict[str, Any]: "review_mode", "verdict", "summary", "findings", ], "properties": { - "schema_version": {"const": 1}, + "schema_version": {"type": "integer", "enum": [1]}, "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, "base_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, "target_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, - "review_mode": {"enum": ["standard", "adversarial"]}, - "verdict": {"enum": ["clean", "findings"]}, - "summary": {"type": "string", "minLength": 1, "maxLength": 20_000}, + "review_mode": { + "type": "string", "enum": ["standard", "adversarial"], + }, + "verdict": {"type": "string", "enum": ["clean", "findings"]}, + "summary": {"type": "string"}, "findings": { "type": "array", "maxItems": MAX_FINDINGS, @@ -53,20 +55,16 @@ def review_output_schema() -> dict[str, Any]: "start_line", "end_line", "evidence", ], "properties": { - "id": {"type": "string", "minLength": 1, "maxLength": 200}, - "severity": {"enum": sorted(FINDING_SEVERITIES)}, - "title": {"type": "string", "minLength": 1, "maxLength": 500}, - "description": { - "type": "string", "minLength": 1, - "maxLength": MAX_TEXT_CHARS, + "id": {"type": "string"}, + "severity": { + "type": "string", "enum": sorted(FINDING_SEVERITIES), }, - "path": {"type": "string", "minLength": 1}, + "title": {"type": "string"}, + "description": {"type": "string"}, + "path": {"type": "string"}, "start_line": {"type": "integer", "minimum": 1}, "end_line": {"type": "integer", "minimum": 1}, - "evidence": { - "type": "string", "minLength": 1, - "maxLength": MAX_TEXT_CHARS, - }, + "evidence": {"type": "string"}, }, }, }, diff --git a/test/core/test_review_workflow.py b/test/core/test_review_workflow.py index 8ec3c3f..a7cc31b 100644 --- a/test/core/test_review_workflow.py +++ b/test/core/test_review_workflow.py @@ -15,6 +15,7 @@ decode_review_document, evaluate_branch_review, make_branch_review_evidence, + review_output_schema, validate_branch_review_evidence, validate_check_results, validate_review_document, @@ -80,6 +81,22 @@ def stream(self, content=""): "full_sha256": hashlib.sha256(encoded).hexdigest(), } + def test_native_output_schema_types_every_property(self): + schema = review_output_schema() + + def visit(value): + if "properties" in value: + self.assertEqual(set(value["required"]), set(value["properties"])) + self.assertFalse(value["additionalProperties"]) + for child in value["properties"].values(): + self.assertIn("type", child) + visit(child) + if isinstance(value.get("items"), dict): + visit(value["items"]) + + visit(schema) + self.assertEqual(schema["properties"]["schema_version"]["enum"], [1]) + def check_result(self, status, *, returncode=None, error_code=None): configured = self.task["checks"][0] return { From 1abbcf92ae3498c58c021c057e977a5b869e9c96 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 17 Sep 2026 21:36:13 +0530 Subject: [PATCH 060/197] docs: record M3 live Codex review --- docs/plans/engineering-team/RESUME.md | 35 +++++----- docs/plans/engineering-team/backlog.json | 9 +++ .../M3-native-codex-review-2026-09-17.json | 64 +++++++++++++++++++ 3 files changed, 91 insertions(+), 17 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/M3-native-codex-review-2026-09-17.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 7dfb3a8..3f97cbe 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -22,7 +22,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne bounded target. `ddb6f51` fixes both: lease authorization now samples time after acquiring the SQLite write transaction, and public cancel resumes an interrupted `recovery_cleanup`. Both have deterministic regressions. -- M3 is in progress through `459ff3f`. `ca55990` adds strict deterministic +- M3 is in progress through `9478796`. `ca55990` adds strict deterministic profile/policy routing, alias binding snapshots, pins/fallbacks, independent reviewer selection and typed pool capacity. `1c7b614` freezes those exact bytes and decisions during public preflight. `cd9a881` resolves base/target @@ -34,15 +34,19 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne `459ff3f` adds the public native Codex reviewer: it freezes the verified CLI, launches one ephemeral read-only app-server turn inside the existing M2 process group, verifies observed model/effort/sandbox identity, accepts only - strict candidate-bound JSON, and records native usage. The current gate is - 167 core tests with `ResourceWarning` promoted to failure plus 202 Bash - assertions. -- Public `branch-review` now reaches the complete native Codex path in offline - protocol tests. Malformed output, denial, disconnect and identity drift all - fail without becoming valid reviews. A bounded real Codex end-to-end proof - is still required before advertising it as ready for real review. M3 also - still needs rich reports for early terminal failures, materialized waiting - handoff reports and the configured headless-lead path. + strict candidate-bound JSON, and records native usage. `e10db94` pins the + compatible bundled Codex 0.153.4 runtime and isolates each review from old + sessions, config, plugins, skills and MCPs while exposing only existing + subscription auth. `9478796` makes the structured-output schema provider + compatible. The current gate is 169 core tests with `ResourceWarning` + promoted to failure plus 202 Bash assertions. +- The bounded real public `branch-review` gate passed at `9478796` with + verified gpt-5.5/low, read-only ephemeral execution, one supported finding, + a passing required check, native-reported usage and all five terminal report + hashes. See the [portable redacted evidence](evidence/M3-native-codex-review-2026-09-17.json). + M3 still needs rich reports for early terminal failures, materialized waiting + handoff reports and the configured headless-lead path before milestone + acceptance. - Current provider readiness is external to M2: Claude CLI is not logged in; Grok CLI authentication expired; Gemini CLI's individual-account path is unsupported and its supported successor is Antigravity; Antigravity is @@ -68,7 +72,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 167 tests passed through the M3 native Codex reviewer path | +| Python core discovery | 169 tests passed through the provider-compatible M3 native Codex reviewer path | | Bash 3.2 regression suite | 10 test files, 202 assertions passed | | Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 5 | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | @@ -79,6 +83,7 @@ Verified at the implementation/evidence checkpoints above: | M3 offline workflow evidence | Strict candidate-bound review/check evaluation and durable separate-worktree execution passed at `a756307` / `97d2c6c` | | M3 host disposition/reporting | Accept/reject/revise, required-check blocking, retry budgets, stale claims, crash resume and five terminal reports passed at `30cf49e` | | M3 native Codex reviewer | Public start, exact identity verification, ephemeral read-only structured output, native usage and four provider-fault classes passed offline at `459ff3f` | +| M3 live public review | Passed at `9478796`; gpt-5.5/low found one supported regression, the required check passed, host acceptance terminalized succeeded and five report hashes were retained | The first two saved-probe invocations failed before `Popen` because of probe-only path/field defects, so neither launched Codex nor consumed a model @@ -98,15 +103,11 @@ are not advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Add a reproducible, opt-in bounded live M3 probe that uses the public - service, keeps raw provider material under `~/.devsquad/private-probes`, and - emits only a redacted receipt/hash. Run it once with the existing - subscription-backed Codex login and retain the evidence. -3. Close the remaining M3 contract gaps: rich terminal reports for failures +2. Close the remaining M3 contract gaps: rich terminal reports for failures before host disposition, materialized waiting handoff reports, and the configured headless-lead path. Rerun the full core/Bash gates and audit M3 against its acceptance section before marking it complete. -4. Keep Claude/Grok/Antigravity probes paused until their normal login or trust +3. Keep Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. They do not block the independent Codex M3 gate. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 7c8f6db..7717f64 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -147,6 +147,15 @@ "artifact": "../../../test/core/test_review_runtime.py", "recorded_at": "2026-09-17T16:55:37+05:30", "availability": "tracked_tests" + }, + { + "kind": "live_acceptance", + "revision": "9478796", + "command_or_action": "Run the public branch-review service with the isolated subscription-backed codex-cli 0.153.4 app-server, gpt-5.5/low, read-only permissions, one required check and host acceptance", + "outcome": "Succeeded with one supported finding, a passing required check, verified observed identity, 53,700 native-reported tokens and all five terminal report hashes; M3 remains in progress for waiting/failure reports, headless lead and final audit", + "artifact": "evidence/M3-native-codex-review-2026-09-17.json", + "recorded_at": "2026-09-17T21:34:40+05:30", + "availability": "portable_redacted" } ], "blocker": null diff --git a/docs/plans/engineering-team/evidence/M3-native-codex-review-2026-09-17.json b/docs/plans/engineering-team/evidence/M3-native-codex-review-2026-09-17.json new file mode 100644 index 0000000..2f27c6c --- /dev/null +++ b/docs/plans/engineering-team/evidence/M3-native-codex-review-2026-09-17.json @@ -0,0 +1,64 @@ +{ + "schema_version": 1, + "milestone": "M3", + "status": "in_progress", + "implementation_revision": "9478796247486ead4e62578cff4035f418534ac7", + "recorded_at": "2026-09-17T21:34:40+05:30", + "evidence": [ + { + "kind": "offline_test", + "command_or_action": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning python3 -m unittest discover -s test/core -p 'test_*.py' -q", + "outcome": "169 core tests passed after the current-version, isolated-auth and provider-compatible structured-output fixes", + "availability": "tracked tests" + }, + { + "kind": "offline_test", + "command_or_action": "bash test/run.sh", + "outcome": "10 Bash test files passed; all 202 Bash 3.2 and optional-jq assertions passed", + "availability": "tracked tests" + }, + { + "kind": "live_public_branch_review", + "command_or_action": "test/core/probes/m3_codex_review.py --run-live --codex-binary /Applications/ChatGPT.app/Contents/Resources/codex --model gpt-5.5 --effort low --timeout 180", + "outcome": "Public Service.start reached a host handoff, the required unit-tests check passed, the native reviewer returned one supported finding, the host accepted the evidence and the run terminalized succeeded with all five required reports", + "availability": "private artifacts retained locally; portable redacted hashes below" + }, + { + "kind": "execution_identity", + "command_or_action": "Freeze and re-observe the selected app-server identity before accepting review evidence", + "outcome": "codex-cli 0.153.4; OpenAI gpt-5.5; low effort; read_only; verified; correlated native thread and turn IDs present; ephemeral thread; isolated temporary CODEX_HOME with only existing subscription auth exposed and apps/plugins/browser/computer-use/multi-agent disabled", + "availability": "portable redacted summary" + }, + { + "kind": "native_usage", + "command_or_action": "Read the correlated thread/tokenUsage/updated event from the successful review turn", + "outcome": "52,271 input tokens; 1,429 output tokens; 53,700 total tokens; source native_reported", + "availability": "portable redacted summary" + } + ], + "candidate_sha256": "2b3bc90aee56082b4df943e4f5877c83ee41fb75a760980d1364e06820efabe7", + "successful_private_probe_receipt_sha256": "a9166be2c2e1a7946d5288a83685d96049e69ccba833a0634d1a4844ad23dd53", + "terminal_report_sha256": { + "artifact-manifest.json": "b2a84aea8419b451bc12ce3c1e9f00be4ecc16b70475366c166fc458af13850e", + "events.jsonl": "24442a87aedc085fcfc038d5382870105a585d8642aa4f8b72e524b21c270529", + "receipt.json": "ba1e13861f4e5b02e6ed22e2118b6a4e125872c7452eb2ebea3007408130f2a3", + "receipt.md": "802594961f53387bfa201fdf4bbb9af871f422e6f835309314f37aa3738b36dd", + "result-receipt.json": "ba1e13861f4e5b02e6ed22e2118b6a4e125872c7452eb2ebea3007408130f2a3" + }, + "failed_probe_evidence": [ + { + "receipt_sha256": "7e096bb8e47cd2081067f535fae4bc71616d1b9c3386927a8024840a82388026", + "outcome": "codex-cli 0.135.0 could no longer decode the current server model catalog after max reasoning metadata was introduced; no valid review evidence was accepted" + }, + { + "receipt_sha256": "bd7d43e7900810307eaba7b5400d218be0427d7204c4222f7b5bd67a3ae1f5fb", + "outcome": "codex-cli 0.153.4 reached the provider but strict structured output rejected the untyped schema_version property with invalid_json_schema; no valid review evidence was accepted" + } + ], + "residual_blockers": [ + "Materialize waiting handoff JSON and Markdown reports", + "Generate rich terminal reports for failures before host disposition", + "Implement and verify the configured headless-lead disposition path", + "Complete an independent M3 acceptance audit before marking the milestone complete" + ] +} From 9dc795a9bb1b0b245c3f01da5ecca7ecc39e173d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 17 Sep 2026 21:46:08 +0530 Subject: [PATCH 061/197] feat: report early M3 review failures --- plugin/core/src/devsquad/reports.py | 277 +++++++++++++++++++++++++ plugin/core/src/devsquad/supervisor.py | 57 ++++- test/core/test_review_runtime.py | 71 ++++++- 3 files changed, 400 insertions(+), 5 deletions(-) diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index a5d60cb..353c9d8 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -20,6 +20,16 @@ "artifact-manifest.json", "result-receipt.json", }) +HANDOFF_REPORT_BASE_NAMES = frozenset({"handoff.json", "handoff.md"}) + + +def handoff_report_names(sequence: int) -> tuple[str, str]: + """Keep the first public names stable and retain later revision packets.""" + if type(sequence) is not int or sequence < 1: + raise ContractError("handoff report sequence is invalid") + if sequence == 1: + return "handoff.json", "handoff.md" + return f"handoff-{sequence}.json", f"handoff-{sequence}.md" def _artifact_projection(artifact: dict[str, Any]) -> dict[str, Any]: @@ -57,6 +67,273 @@ def _event_export(events: list[dict[str, Any]]) -> tuple[bytes, int, int]: return content, last_cursor, last_version +def _unreferenced_artifact_projection(artifact: dict[str, Any]) -> dict[str, Any]: + """Project an artifact before the atomic import has assigned its database id.""" + required = {"name", "sha256", "byte_size"} + if not isinstance(artifact, dict) or not required <= set(artifact): + raise ContractError("unreferenced report artifact projection is invalid") + name, digest, size = ( + artifact["name"], artifact["sha256"], artifact["byte_size"], + ) + if (not isinstance(name, str) or not name + or not isinstance(digest, str) or len(digest) != 64 + or any(character not in "0123456789abcdef" for character in digest) + or type(size) is not int or size < 0): + raise ContractError("unreferenced report artifact values are invalid") + return {"id": None, "name": name, "sha256": digest, "byte_size": size} + + +def _contents_with_manifest( + run_id: str, + receipt: dict[str, Any], + markdown: str, + events_content: bytes, + projected_artifacts: list[dict[str, Any]], +) -> dict[str, bytes]: + receipt_content = (canonical_json(receipt) + "\n").encode() + contents = { + "receipt.json": receipt_content, + "receipt.md": markdown.encode(), + "events.jsonl": events_content, + "result-receipt.json": receipt_content, + } + manifest_entries = list(projected_artifacts) + for name, content in contents.items(): + manifest_entries.append({ + "id": None, + "name": name, + "sha256": hashlib.sha256(content).hexdigest(), + "byte_size": len(content), + }) + manifest = { + "schema_version": 1, + "run_id": run_id, + "candidate_sha256": receipt["candidate"]["sha256"], + "artifacts": manifest_entries, + "self_excluded": True, + } + contents["artifact-manifest.json"] = ( + canonical_json(manifest) + "\n" + ).encode() + if set(contents) != TERMINAL_REPORT_NAMES: + raise ContractError("terminal report set is incomplete") + return contents + + +def build_handoff_reports( + *, + run_id: str, + handoff_id: str, + sequence: int, + packet: dict[str, Any], + packet_sha256: str, + created_at: str, +) -> dict[str, bytes]: + """Build the portable JSON and Markdown view of an open host handoff.""" + if not all(isinstance(value, str) and value for value in ( + run_id, handoff_id, packet_sha256, created_at, + )): + raise ContractError("handoff report identity is invalid") + packet_json = canonical_json(packet) + if hashlib.sha256(packet_json.encode()).hexdigest() != packet_sha256: + raise ContractError("handoff report packet hash is invalid") + json_name, markdown_name = handoff_report_names(sequence) + report = { + "schema_version": 1, + "run_id": run_id, + "workflow": "branch-review", + "state": "awaiting_host", + "created_at": created_at, + "handoff_id": handoff_id, + "sequence": sequence, + "packet_sha256": packet_sha256, + "candidate": { + "sha256": packet.get("candidate_sha256"), + "base_oid": packet.get("base_oid"), + "target_oid": packet.get("target_oid"), + }, + "packet": packet, + "next_action": "claim_handoff", + } + review = packet.get("review") if isinstance(packet.get("review"), dict) else {} + lines = [ + "# DevSquad branch review handoff", + "", + f"- Run: `{run_id}`", + f"- Handoff: `{handoff_id}`", + f"- Sequence: `{sequence}`", + f"- Candidate: `{packet.get('candidate_sha256')}`", + f"- Review verdict: `{review.get('verdict', 'unavailable')}`", + "- Next action: claim this saved handoff and submit one disposition.", + "", + "## Review", + "", + str(review.get("summary") or "No review summary was supplied."), + "", + "## Findings", + "", + ] + findings = review.get("findings") + if isinstance(findings, list) and findings: + for finding in findings: + lines.append( + f"- **{str(finding.get('severity', 'unknown')).upper()} — " + f"{finding.get('title', 'Untitled finding')}** " + f"(`{finding.get('path', '?')}:{finding.get('start_line', '?')}`)" + ) + else: + lines.append("- No supported findings were reported.") + lines.extend(["", "## Instructions", "", str(packet.get("instructions") or "")]) + return { + json_name: (canonical_json(report) + "\n").encode(), + markdown_name: ("\n".join(lines) + "\n").encode(), + } + + +def build_early_terminal_reports( + *, + run_id: str, + state: str, + task: dict[str, Any], + snapshot: dict[str, Any] | None, + run_artifacts: list[dict[str, Any]], + events: list[dict[str, Any]], + completed_at: str, + phase: str, + error: dict[str, Any] | None, + attempt: dict[str, Any] | None = None, +) -> dict[str, bytes]: + """Build the M3 report set when no valid handoff/lead decision exists.""" + if not isinstance(run_id, str) or not run_id: + raise ContractError("report run id is invalid") + if state not in {"failed", "cancelled"}: + raise ContractError("early terminal report state is invalid") + if not isinstance(task, dict) or task.get("workflow") != "branch-review": + raise ContractError("early terminal report task is invalid") + if snapshot is not None and not isinstance(snapshot, dict): + raise ContractError("early terminal report snapshot is invalid") + if not isinstance(completed_at, str) or not completed_at or not phase: + raise ContractError("early terminal report completion is invalid") + if error is not None and not isinstance(error, dict): + raise ContractError("early terminal report error is invalid") + + frozen = snapshot or {} + workspace = frozen.get("workspace") + workspace = workspace if isinstance(workspace, dict) else {} + projected = [ + _unreferenced_artifact_projection(artifact) for artifact in run_artifacts + ] + events_content, through_cursor, through_version = _event_export(events) + attempt_projection = None + if attempt is not None: + if not isinstance(attempt, dict) or not isinstance(attempt.get("id"), str): + raise ContractError("early terminal report attempt is invalid") + attempt_projection = { + "id": attempt["id"], + "role": "reviewer", + "status": state, + "returncode": attempt.get("returncode"), + "cancelled": bool(attempt.get("cancelled", state == "cancelled")), + "timed_out": bool(attempt.get("timed_out", False)), + "selected_profile": ( + frozen.get("routing", {}).get("roles", {}).get("reviewer", {}).get( + "selected" + ) + if isinstance(frozen.get("routing"), dict) else None + ), + "observed_identity": None, + "worker_invocations": 1, + "native_model_requests": None, + "usage": { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + "output_artifacts": projected, + "error": error, + } + criteria = [{ + "id": criterion.get("id"), + "description": criterion.get("description"), + "evidence_kind": criterion.get("evidence_kind"), + "status": "not_evaluated", + "evidence_refs": [], + } for criterion in task.get("acceptance", []) if isinstance(criterion, dict)] + receipt = { + "schema_version": 1, + "run_id": run_id, + "workflow": "branch-review", + "state": state, + "phase": phase, + "completed_at": completed_at, + "candidate": { + "sha256": workspace.get("candidate_sha256"), + "base_oid": workspace.get("base_oid", frozen.get("base_oid")), + "target_oid": workspace.get("target_oid", frozen.get("target_oid")), + }, + "routing": frozen.get("routing"), + "review": None, + "checks": [], + "evaluation": None, + "criteria": criteria, + "attempts": [attempt_projection] if attempt_projection else [], + "dispositions": [], + "lead": { + "mode": task.get("lead", {}).get("mode"), + "status": "not_reached", + "disposition": None, + "reason": None, + "usage": { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + }, + "accounting": { + "worker_invocations": 1 if attempt_projection else 0, + "native_model_requests": None if attempt_projection else 0, + "attempt_usage": ( + [attempt_projection["usage"]] if attempt_projection else [] + ), + "host_usage_measured": False, + }, + "artifacts": projected, + "evidence_artifacts": [], + "events_export": { + "through_cursor": through_cursor, + "through_run_version": through_version, + "includes_terminal_event": False, + "excludes_terminal_report_artifact_events": True, + }, + "limitations": [ + "No valid review handoff was produced, so lead disposition was not reached." + ], + "error": error, + } + lines = [ + "# DevSquad branch review", + "", + f"- Run: `{run_id}`", + f"- State: `{state}`", + f"- Phase: `{phase}`", + "- Lead disposition: `not_reached`", + "", + "## Failure", + "", + str((error or {}).get("message") or (error or {}).get("error") or "Cancelled."), + "", + "## Recovery", + "", + "Start a new run with a new idempotency key after correcting the recorded error.", + "", + ] + return _contents_with_manifest( + run_id, receipt, "\n".join(lines), events_content, projected, + ) + + def _decision(value: Any) -> dict[str, Any]: fields = { "schema_version", "submission_id", "submission_hash", "disposition", diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 1db6566..bed7746 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -4,6 +4,7 @@ from dataclasses import dataclass import ctypes +from datetime import datetime, timezone import hashlib import os import signal @@ -17,6 +18,7 @@ import json from .contracts import ContractError, LaunchSpec +from .reports import build_early_terminal_reports from .store import AttemptReservation, ConflictError, Store, canonical_json from .workflows import decode_branch_review_evidence @@ -417,15 +419,64 @@ def import_durable(self, run_id: str) -> str: semantic_error=str(exc) receipt["error"]="WORKFLOW_OUTPUT_INVALID" receipt["message"]=semantic_error - receipt_bytes=canonical_json(receipt).encode() - path,digest,size=self.store.finalize_artifact(run_id,"result-receipt.json",receipt_bytes) - artifacts.append({"name":"result-receipt.json","path":path,"sha256":digest,"byte_size":size}) terminal="cancelled" if receipt["cancelled"] else ("failed" if receipt["timed_out"] or receipt["returncode"]!=0 or semantic_error else "succeeded") payload={"returncode":receipt["returncode"],"receipt":"result-receipt.json"} if receipt["timed_out"]: payload["error"]="TIMEOUT" if semantic_error: payload["error"]="WORKFLOW_OUTPUT_INVALID" payload["message"]=semantic_error + if workflow_review: + if receipt["cancelled"]: + report_error = None + elif receipt["timed_out"]: + report_error = { + "error": "TIMEOUT", + "message": "branch review worker exceeded its deadline", + } + elif semantic_error: + report_error = { + "error": "WORKFLOW_OUTPUT_INVALID", + "message": semantic_error, + } + else: + report_error = { + "error": "REVIEW_WORKER_FAILED", + "message": "branch review worker exited before producing a valid handoff", + "returncode": receipt["returncode"], + } + payload.update(report_error) + reports = build_early_terminal_reports( + run_id=run_id, + state=terminal, + task=snapshot["task"], + snapshot=snapshot, + run_artifacts=artifacts, + events=self.store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + phase="reviewer", + error=report_error, + attempt={ + "id": attempt["id"], + "returncode": receipt["returncode"], + "cancelled": receipt["cancelled"], + "timed_out": receipt["timed_out"], + }, + ) + for name in sorted(reports): + content = reports[name] + path,digest,size=self.store.finalize_artifact( + run_id, name, content, + ) + artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + else: + receipt_bytes=canonical_json(receipt).encode() + path,digest,size=self.store.finalize_artifact(run_id,"result-receipt.json",receipt_bytes) + artifacts.append({"name":"result-receipt.json","path":path,"sha256":digest,"byte_size":size}) return self.store.commit_durable_import( run_id,attempt["attempt_token"],artifacts,metadata,terminal,payload, ) diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index 6b13a71..ff04b8c 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -15,6 +15,7 @@ sys.path.insert(0, str(ROOT / "plugin/core/src")) from devsquad.contracts import ContractError +from devsquad.reports import TERMINAL_REPORT_NAMES, build_handoff_reports from devsquad.service import Service from devsquad.store import ConflictError, Store, request_hash from devsquad_test_fixtures import branch_review_routing_documents @@ -522,9 +523,75 @@ def test_native_codex_faults_never_become_valid_reviews(self): self.assertIsNone(failed["handoff"]) result = self.service.result(started["run_id"]) self.assertTrue(result["ready"]) - self.assertNotIn( - "receipt.json", {artifact["name"] for artifact in result["artifacts"]}, + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["state"], "failed") + self.assertEqual(receipt["lead"]["status"], "not_reached") + self.assertEqual(receipt["error"]["error"], "REVIEW_WORKER_FAILED") + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + self.assertEqual(receipt["attempts"][0]["status"], "failed") + self.assertEqual( + {item["name"] for item in receipt["attempts"][0]["output_artifacts"]}, + { + f"{receipt['attempts'][0]['id']}.stdout", + f"{receipt['attempts'][0]['id']}.stderr", + }, ) + self.assertEqual( + Path(artifacts["receipt.json"]["path"]).read_bytes(), + Path(artifacts["result-receipt.json"]["path"]).read_bytes(), + ) + manifest = json.loads( + Path(artifacts["artifact-manifest.json"]["path"]).read_text() + ) + for entry in manifest["artifacts"]: + saved = Path(artifacts[entry["name"]]["path"]).read_bytes() + self.assertEqual(hashlib.sha256(saved).hexdigest(), entry["sha256"]) + self.assertEqual(len(saved), entry["byte_size"]) + + def test_check_worker_failure_before_handoff_gets_full_terminal_reports(self): + self.task["checks"][0]["cwd"] = "missing-check-directory" + started = self.service.start( + self.task, + "check-worker-failure", + _internal_review_fixture=self.fixture, + ) + failed = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(failed["state"], "failed") + self.assertIsNone(failed["handoff"]) + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "REVIEW_WORKER_FAILED") + self.assertIsNone(receipt["review"]) + self.assertEqual(receipt["checks"], []) + self.assertEqual(receipt["lead"]["status"], "not_reached") + + def test_waiting_handoff_report_builder_binds_packet_and_hash(self): + run_id, _ = self.start_waiting("handoff-report-builder") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + handoff = store.handoff_snapshot(run_id) + self.assertIsNotNone(handoff) + reports = build_handoff_reports( + run_id=run_id, + handoff_id=handoff.handoff_id, + sequence=handoff.sequence, + packet=handoff.packet, + packet_sha256=handoff.packet_sha256, + created_at=handoff.created_at, + ) + finally: + store.close() + self.assertEqual(set(reports), {"handoff.json", "handoff.md"}) + document = json.loads(reports["handoff.json"]) + self.assertEqual(document["state"], "awaiting_host") + self.assertEqual(document["packet"], handoff.packet) + self.assertEqual(document["packet_sha256"], handoff.packet_sha256) + self.assertIn(b"claim this saved handoff", reports["handoff.md"]) def test_invalid_internal_review_fails_before_launch_with_a_receipt(self): invalid = dict(self.fixture, verdict="clean") From 8f3fa56a22069d3f6230772a9d170b0d9cfb4296 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 18 Sep 2026 13:56:01 +0530 Subject: [PATCH 062/197] WIP checkpoint: feat: add durable headless review leadership (2026-09-18 13:56) --- plugin/core/src/devsquad/codex_lead_worker.py | 272 ++++++++++++++++++ .../core/src/devsquad/codex_review_worker.py | 59 ++-- plugin/core/src/devsquad/detached.py | 52 +++- plugin/core/src/devsquad/lead_worker.py | 42 +++ .../devsquad/migrations/006_attempt_roles.sql | 1 + plugin/core/src/devsquad/reports.py | 77 +++-- plugin/core/src/devsquad/service.py | 246 +++++++++++++++- plugin/core/src/devsquad/store.py | 233 ++++++++++++++- plugin/core/src/devsquad/supervisor.py | 91 +++++- plugin/core/src/devsquad/workflows.py | 240 ++++++++++++++++ test/core/fakes/codex_review_cli.py | 28 +- test/core/test_cli.py | 6 +- test/core/test_handoff_store.py | 6 +- test/core/test_m2_cross_process.py | 6 +- test/core/test_review_runtime.py | 210 +++++++++++++- test/core/test_service.py | 19 +- test/core/test_store.py | 12 +- 17 files changed, 1498 insertions(+), 102 deletions(-) create mode 100644 plugin/core/src/devsquad/codex_lead_worker.py create mode 100644 plugin/core/src/devsquad/lead_worker.py create mode 100644 plugin/core/src/devsquad/migrations/006_attempt_roles.sql diff --git a/plugin/core/src/devsquad/codex_lead_worker.py b/plugin/core/src/devsquad/codex_lead_worker.py new file mode 100644 index 0000000..d4c72db --- /dev/null +++ b/plugin/core/src/devsquad/codex_lead_worker.py @@ -0,0 +1,272 @@ +"""Drive one frozen read-only Codex headless-lead turn inside the M2 worker.""" + +from __future__ import annotations + +import hashlib +import io +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import time +from typing import Any + +from .catalog import normalize_models, verified_efforts +from .codex_protocol import ( + JsonLinePeer, + NativeTurnState, + discover_models, + initialize_request, + initialized_notification, + receive_response, + thread_start_request, + turn_start_request, +) +from .codex_review_worker import ( + MAX_PROTOCOL_EVENT_BYTES, + MAX_PROTOCOL_EVENTS, + MAX_SERVER_STDERR_BYTES, + _isolated_codex_environment, + _remaining, + _response_result, + _stop_server, + _usage, + _validated_adapter, + freeze_codex_role, +) +from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported +from .store import canonical_json +from .supervisor import BoundedDrain +from .workflows import ( + MAX_EVIDENCE_BYTES, + MAX_LEAD_BYTES, + build_lead_prompt, + decode_headless_lead_choice, + lead_output_schema, + make_headless_lead_evidence, +) + + +def freeze_codex_lead(selected: dict[str, Any]) -> dict[str, Any]: + return freeze_codex_role( + selected, role="lead", output_schema=lead_output_schema(), + ) + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("workflow snapshot must be an object") + adapter, profile = _validated_adapter( + snapshot, + role="lead", + adapter_key="lead_adapter", + output_schema=lead_output_schema(), + ) + handoff = snapshot.get("headless_handoff") + if not isinstance(handoff, dict) or not isinstance(handoff.get("packet"), dict): + raise ContractError("headless lead handoff is missing") + binary = Path(adapter["binary"]) + try: + resolved = binary.resolve(strict=True) + except OSError as exc: + raise CapabilityUnavailable("frozen Codex executable is missing") from exc + if (resolved != binary + or hashlib.sha256(binary.read_bytes()).hexdigest() != adapter["binary_sha256"]): + raise CapabilityUnavailable("frozen Codex executable changed after preflight") + try: + version = subprocess.run( + [str(binary), "--version"], text=True, capture_output=True, + timeout=3, check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise CapabilityUnavailable( + "frozen Codex version could not be re-observed" + ) from exc + if version.returncode != 0 or version.stdout.strip() != adapter["harness_version"]: + raise CapabilityUnavailable("Codex version changed after preflight") + + effort = profile["effort"]["value"] + model = profile["model_id"] + review_root = Path(snapshot["workspace"]["path"]).resolve(strict=True) + argv = [ + str(binary), + "-c", f'model="{model}"', + "-c", f'model_reasoning_effort="{effort}"', + "--disable", "apps", + "--disable", "plugins", + "--disable", "browser_use", + "--disable", "computer_use", + "--disable", "multi_agent", + "--enable", "skip_host_skill_discovery", + "app-server", "--listen", "stdio://", + ] + process: subprocess.Popen[bytes] | None = None + writer: io.TextIOWrapper | None = None + stderr: BoundedDrain | None = None + codex_home: tempfile.TemporaryDirectory | None = None + protocol_events: list[dict[str, Any]] = [] + protocol_bytes = 0 + deadline = time.monotonic() + snapshot["task"]["budget"]["wall_seconds"] + + def record(message: dict[str, Any]) -> None: + nonlocal protocol_bytes + encoded = canonical_json(message).encode() + protocol_bytes += len(encoded) + if (len(protocol_events) >= MAX_PROTOCOL_EVENTS + or protocol_bytes > MAX_PROTOCOL_EVENT_BYTES): + raise ContractError("native Codex protocol evidence exceeds its bound") + protocol_events.append(message) + + try: + codex_home, environment = _isolated_codex_environment(adapter["auth_file"]) + process = subprocess.Popen( + argv, + cwd=review_root, + env=environment, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + start_new_session=False, + close_fds=True, + ) + assert process.stdin is not None and process.stdout is not None and process.stderr is not None + writer = io.TextIOWrapper(process.stdin, encoding="utf-8", write_through=True) + peer = JsonLinePeer(process.stdout, writer) + stderr = BoundedDrain(process.stderr, MAX_SERVER_STDERR_BYTES) + stderr.start() + + peer.send(initialize_request(1)) + _response_result(receive_response( + peer, 1, timeout_seconds=_remaining(deadline), on_notification=record, + ), "Codex initialize") + peer.send(initialized_notification()) + models = discover_models( + peer, first_request_id=10, timeout_seconds=_remaining(deadline), + ) + catalog = { + "complete": True, + "harness": "codex", + "harness_version": adapter["harness_version"], + "models": normalize_models("codex", adapter["harness_version"], models), + } + if effort not in verified_efforts( + catalog, + harness="codex", + version=adapter["harness_version"], + model_id=model, + ): + raise ProfileUnsupported( + f"unsupported or unverified effort {effort!r} for codex model {model!r}" + ) + + peer.send(thread_start_request( + 100, cwd=str(review_root), model=model, permission="read_only", + ephemeral=True, + )) + thread_result = _response_result(receive_response( + peer, 100, timeout_seconds=_remaining(deadline), on_notification=record, + ), "Codex thread/start") + thread = thread_result.get("thread") + thread_id = thread.get("id") if isinstance(thread, dict) else None + reported_version = thread.get("cliVersion") if isinstance(thread, dict) else None + if not isinstance(thread_id, str) or not thread_id: + raise ContractError("Codex thread/start returned no thread id") + if reported_version and f"codex-cli {reported_version}" != adapter["harness_version"]: + raise ContractError("Codex thread reported a different harness version") + reported_cwd = thread_result.get("cwd") + expected_policy = {"type": "readOnly", "networkAccess": False} + if (not isinstance(reported_cwd, str) or not reported_cwd + or thread_result.get("model") != model + or thread_result.get("reasoningEffort") != effort + or thread_result.get("modelProvider") != adapter["model_provider"] + or thread_result.get("approvalPolicy") != "never" + or thread_result.get("sandbox") != expected_policy + or Path(reported_cwd).resolve() != review_root): + raise ContractError("Codex thread did not preserve the frozen execution identity") + + prompt = build_lead_prompt(snapshot["task"], handoff["packet"]) + peer.send(turn_start_request( + 101, + thread_id=thread_id, + prompt=prompt, + model=model, + effort=effort, + cwd=str(review_root), + permission="read_only", + output_schema=lead_output_schema(), + )) + turn_result = _response_result(receive_response( + peer, 101, timeout_seconds=_remaining(deadline), on_notification=record, + ), "Codex turn/start") + turn = turn_result.get("turn") + turn_id = turn.get("id") if isinstance(turn, dict) else None + if not isinstance(turn_id, str) or not turn_id: + raise ContractError("Codex turn/start returned no turn id") + state = NativeTurnState(thread_id=thread_id, turn_id=turn_id) + for event in protocol_events: + state.consume(event) + output_bytes = sum(len(part.encode()) for part in state.output) + while not state.terminal: + message = peer.receive(_remaining(deadline)) + if "id" in message and "method" in message: + raise ContractError("native Codex requested an unsupported host action") + record(message) + prior = len(state.output) + state.consume(message) + output_bytes += sum(len(part.encode()) for part in state.output[prior:]) + if output_bytes > MAX_LEAD_BYTES: + raise ContractError("native Codex lead output exceeds its byte limit") + if state.terminal_status != "completed": + detail = ( + canonical_json(state.error)[:2000] + if state.error is not None else "no provider error was reported" + ) + raise ContractError( + "native Codex lead did not complete successfully: " + f"{state.terminal_status}; {detail}" + ) + choice = decode_headless_lead_choice( + "".join(state.output).strip(), handoff["packet"], + ) + usage = _usage(protocol_events, thread_id, turn_id) + finally: + if process is not None: + _stop_server(process, writer, stderr) + if codex_home is not None: + codex_home.cleanup() + + observed_identity = { + "harness": "codex", + "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], + "model_id": model, + "effort": effort, + "permission_policy": "read_only", + "verification": "verified", + } + return make_headless_lead_evidence( + snapshot, + handoff, + choice, + observed_identity=observed_identity, + native_ids={"thread_id": thread_id, "turn_id": turn_id}, + native_model_requests=None, + usage=usage, + ) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_EVIDENCE_BYTES + 1) + if len(payload) > MAX_EVIDENCE_BYTES: + raise ContractError("workflow snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("workflow snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/codex_review_worker.py b/plugin/core/src/devsquad/codex_review_worker.py index 7536511..581f735 100644 --- a/plugin/core/src/devsquad/codex_review_worker.py +++ b/plugin/core/src/devsquad/codex_review_worker.py @@ -76,27 +76,32 @@ def _subscription_auth_file(value: str | None = None) -> Path: return auth_file -def freeze_codex_reviewer(selected: dict[str, Any]) -> dict[str, Any]: - """Resolve and verify the non-model Codex launch identity during preflight.""" +def freeze_codex_role( + selected: dict[str, Any], + *, + role: str, + output_schema: dict[str, Any], +) -> dict[str, Any]: + """Resolve and verify one read-only Codex role during preflight.""" from .adapters import AdapterManifest, harness_version if not isinstance(selected, dict) or not isinstance(selected.get("profile"), dict): - raise ContractError("frozen reviewer selection is invalid") + raise ContractError(f"frozen {role} selection is invalid") profile = selected["profile"] if profile.get("harness") != "codex": raise CapabilityUnavailable( - f"selected reviewer harness is not implemented for M3: {profile.get('harness')}" + f"selected {role} harness is not implemented for M3: {profile.get('harness')}" ) if profile.get("permission_policy") != "read_only": - raise ProfileUnsupported("Codex branch review requires read_only permission") + raise ProfileUnsupported(f"Codex {role} requires read_only permission") if set(profile.get("required_tools", [])) - {"read"}: - raise ProfileUnsupported("Codex branch review profile requests unsupported tools") + raise ProfileUnsupported(f"Codex {role} profile requests unsupported tools") effort = profile.get("effort") if (not isinstance(effort, dict) or effort.get("transport") != "native" or not isinstance(effort.get("value"), str) or not effort["value"]): - raise ProfileUnsupported("Codex branch review requires an explicit native effort") + raise ProfileUnsupported(f"Codex {role} requires an explicit native effort") if not isinstance(profile.get("model_id"), str) or not profile["model_id"]: - raise ProfileUnsupported("Codex branch review requires an exact model id") + raise ProfileUnsupported(f"Codex {role} requires an exact model id") manifest = AdapterManifest.load(_adapter_manifest_path()) if manifest.name != "codex" or manifest.transport != "native_protocol": @@ -111,7 +116,7 @@ def freeze_codex_reviewer(selected: dict[str, Any]) -> dict[str, Any]: if version not in manifest.verified_versions: raise ProfileUnsupported(f"unverified Codex app-server version: {version}") content = binary.read_bytes() - schema_hash = hashlib.sha256(canonical_json(review_output_schema()).encode()).hexdigest() + schema_hash = hashlib.sha256(canonical_json(output_schema).encode()).hexdigest() return { "schema_version": 1, "harness": "codex", @@ -125,37 +130,51 @@ def freeze_codex_reviewer(selected: dict[str, Any]) -> dict[str, Any]: } -def _validated_adapter(snapshot: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any]]: - adapter = snapshot.get("review_adapter") +def freeze_codex_reviewer(selected: dict[str, Any]) -> dict[str, Any]: + """Resolve and verify the non-model Codex reviewer identity.""" + return freeze_codex_role( + selected, role="reviewer", output_schema=review_output_schema(), + ) + + +def _validated_adapter( + snapshot: dict[str, Any], + *, + role: str = "reviewer", + adapter_key: str = "review_adapter", + output_schema: dict[str, Any] | None = None, +) -> tuple[dict[str, Any], dict[str, Any]]: + adapter = snapshot.get(adapter_key) try: - selected = snapshot["routing"]["roles"]["reviewer"]["selected"] + selected = snapshot["routing"]["roles"][role]["selected"] profile = selected["profile"] except (KeyError, TypeError) as exc: - raise ContractError("frozen Codex reviewer selection is missing") from exc + raise ContractError(f"frozen Codex {role} selection is missing") from exc if not isinstance(adapter, dict) or set(adapter) != ADAPTER_FIELDS: - raise ContractError("frozen Codex review adapter fields are invalid") + raise ContractError(f"frozen Codex {role} adapter fields are invalid") if (adapter["schema_version"] != 1 or type(adapter["schema_version"]) is not int or adapter["harness"] != "codex" or adapter["transport"] != "native_protocol" or adapter["model_provider"] != "openai"): - raise ContractError("frozen Codex review adapter identity is invalid") + raise ContractError(f"frozen Codex {role} adapter identity is invalid") for field in ( "binary", "binary_sha256", "harness_version", "output_schema_sha256", "auth_file", ): if not isinstance(adapter[field], str) or not adapter[field]: - raise ContractError("frozen Codex review adapter value is invalid") + raise ContractError(f"frozen Codex {role} adapter value is invalid") if (len(adapter["binary_sha256"]) != 64 or len(adapter["output_schema_sha256"]) != 64): - raise ContractError("frozen Codex review adapter hash is invalid") + raise ContractError(f"frozen Codex {role} adapter hash is invalid") if not Path(adapter["auth_file"]).is_absolute(): raise ContractError("frozen Codex auth path is not absolute") if (not isinstance(profile, dict) or profile.get("harness") != "codex" or profile.get("permission_policy") != "read_only"): - raise ContractError("frozen profile is not a read-only Codex reviewer") - schema_hash = hashlib.sha256(canonical_json(review_output_schema()).encode()).hexdigest() + raise ContractError(f"frozen profile is not a read-only Codex {role}") + schema = review_output_schema() if output_schema is None else output_schema + schema_hash = hashlib.sha256(canonical_json(schema).encode()).hexdigest() if schema_hash != adapter["output_schema_sha256"]: - raise ContractError("frozen review output schema changed") + raise ContractError(f"frozen {role} output schema changed") return adapter, profile diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index d2d729f..bd44104 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -6,6 +6,7 @@ import sys from .contracts import ExecutionIdentity, LaunchSpec +from .service import Service from .store import ConflictError, Store, canonical_json from .supervisor import Supervisor @@ -28,9 +29,21 @@ def main(argv=None): "DEVSQUAD_DELEGATION_DEPTH": "1", } stdin_path = None - if "internal_review_fixture" in snapshot or "review_adapter" in snapshot: - selected = snapshot["routing"]["roles"]["reviewer"]["selected"]["profile"] - adapter = snapshot.get("review_adapter") + adapter = None + handoff = store.handoff_snapshot(args.run_id) + headless_lead = ( + snapshot["task"]["lead"]["mode"] == "headless" + and handoff is not None + and handoff.status == "open" + ) + role = "lead" if headless_lead else "reviewer" + workflow_role = ( + "internal_review_fixture" in snapshot or "review_adapter" in snapshot + ) + if workflow_role: + selected = snapshot["routing"]["roles"][role]["selected"]["profile"] + adapter_key = "lead_adapter" if headless_lead else "review_adapter" + adapter = snapshot.get(adapter_key) identity = ExecutionIdentity( selected["harness"], adapter["harness_version"] if adapter else "fixture", @@ -44,14 +57,22 @@ def main(argv=None): "verified" if adapter else "unknown", ) module = ( - "devsquad.codex_review_worker" - if adapter else "devsquad.review_worker" + ("devsquad.codex_lead_worker" if adapter else "devsquad.lead_worker") + if headless_lead + else ("devsquad.codex_review_worker" if adapter else "devsquad.review_worker") ) command = [sys.executable, "-P", "-m", module] + worker_snapshot = dict(snapshot) + if headless_lead: + worker_snapshot["headless_handoff"] = { + "handoff_id": handoff.handoff_id, + "packet": handoff.packet, + "packet_sha256": handoff.packet_sha256, + } input_path, _, _ = store.finalize_artifact( args.run_id, - "workflow-input.json", - canonical_json(snapshot).encode(), + "lead-workflow-input.json" if headless_lead else "workflow-input.json", + canonical_json(worker_snapshot).encode(), ) stdin_path = str(input_path) else: @@ -62,7 +83,7 @@ def main(argv=None): spec = LaunchSpec( 1, identity.harness, - "native_protocol" if "review_adapter" in snapshot else "cli_exec", + "native_protocol" if adapter else "cli_exec", tuple(command), run["worktree_path"], stdin_path, @@ -71,9 +92,18 @@ def main(argv=None): environment, ) supervisor = Supervisor(store) - try: handle = supervisor.launch_durable(args.run_id, args.expected_version, spec, f"daemon:{os.getpid()}", args.package_digest) + try: handle = supervisor.launch_durable(args.run_id, args.expected_version, spec, f"daemon:{os.getpid()}", args.package_digest, role=role if workflow_role else "worker") except ConflictError: return 0 - return 0 if supervisor.wait_durable(handle, spec.timeout_seconds) == 0 else 1 - finally: store.close() + returncode = supervisor.wait_durable(handle, spec.timeout_seconds) + current = store.run(args.run_id) + finally: + store.close() + if (current["state"] == "awaiting_host" + and snapshot["task"]["lead"]["mode"] == "headless"): + try: + Service(Path(args.database).parent).resume(args.run_id) + except ConflictError: + pass + return 0 if returncode == 0 else 1 if __name__ == "__main__": raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/lead_worker.py b/plugin/core/src/devsquad/lead_worker.py new file mode 100644 index 0000000..0793eac --- /dev/null +++ b/plugin/core/src/devsquad/lead_worker.py @@ -0,0 +1,42 @@ +"""Run one frozen offline headless-lead fixture for branch-review tests.""" + +from __future__ import annotations + +import json +import sys +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .workflows import MAX_EVIDENCE_BYTES, make_headless_lead_evidence + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("workflow snapshot must be an object") + fixture = snapshot.get("internal_lead_fixture") + handoff = snapshot.get("headless_handoff") + if not isinstance(fixture, dict) or not isinstance(handoff, dict): + raise ContractError("offline headless lead snapshot is incomplete") + choice = { + "schema_version": 1, + "candidate_sha256": handoff["packet"]["candidate_sha256"], + **fixture, + } + return make_headless_lead_evidence(snapshot, handoff, choice) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_EVIDENCE_BYTES + 1) + if len(payload) > MAX_EVIDENCE_BYTES: + raise ContractError("workflow snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("workflow snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/migrations/006_attempt_roles.sql b/plugin/core/src/devsquad/migrations/006_attempt_roles.sql new file mode 100644 index 0000000..b286484 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/006_attempt_roles.sql @@ -0,0 +1 @@ +ALTER TABLE attempts ADD COLUMN role TEXT NOT NULL DEFAULT 'worker'; diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index 353c9d8..d3ffc67 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -202,6 +202,7 @@ def build_early_terminal_reports( phase: str, error: dict[str, Any] | None, attempt: dict[str, Any] | None = None, + prior_attempts: list[dict[str, Any]] | None = None, ) -> dict[str, bytes]: """Build the M3 report set when no valid handoff/lead decision exists.""" if not isinstance(run_id, str) or not run_id: @@ -225,12 +226,16 @@ def build_early_terminal_reports( ] events_content, through_cursor, through_version = _event_export(events) attempt_projection = None + prior = [] if prior_attempts is None else prior_attempts + if not isinstance(prior, list) or not all( + isinstance(item, dict) for item in prior): + raise ContractError("early terminal prior attempts are invalid") if attempt is not None: if not isinstance(attempt, dict) or not isinstance(attempt.get("id"), str): raise ContractError("early terminal report attempt is invalid") attempt_projection = { "id": attempt["id"], - "role": "reviewer", + "role": attempt.get("role", "reviewer"), "status": state, "returncode": attempt.get("returncode"), "cancelled": bool(attempt.get("cancelled", state == "cancelled")), @@ -277,11 +282,11 @@ def build_early_terminal_reports( "checks": [], "evaluation": None, "criteria": criteria, - "attempts": [attempt_projection] if attempt_projection else [], + "attempts": prior + ([attempt_projection] if attempt_projection else []), "dispositions": [], "lead": { "mode": task.get("lead", {}).get("mode"), - "status": "not_reached", + "status": "failed" if phase == "lead" else "not_reached", "disposition": None, "reason": None, "usage": { @@ -292,10 +297,13 @@ def build_early_terminal_reports( }, }, "accounting": { - "worker_invocations": 1 if attempt_projection else 0, - "native_model_requests": None if attempt_projection else 0, + "worker_invocations": sum( + item.get("worker_invocations", 0) for item in prior + ) + (1 if attempt_projection else 0), + "native_model_requests": None if (prior or attempt_projection) else 0, "attempt_usage": ( - [attempt_projection["usage"]] if attempt_projection else [] + [item["usage"] for item in prior] + + ([attempt_projection["usage"]] if attempt_projection else []) ), "host_usage_measured": False, }, @@ -307,9 +315,11 @@ def build_early_terminal_reports( "includes_terminal_event": False, "excludes_terminal_report_artifact_events": True, }, - "limitations": [ - "No valid review handoff was produced, so lead disposition was not reached." - ], + "limitations": [( + "The headless lead failed before a valid disposition was recorded." + if phase == "lead" + else "No valid review handoff was produced, so lead disposition was not reached." + )], "error": error, } lines = [ @@ -477,6 +487,7 @@ def build_terminal_reports( events: list[dict[str, Any]], completed_at: str, error: dict[str, Any] | None = None, + headless_leads: list[dict[str, Any]] | None = None, ) -> dict[str, bytes]: if not isinstance(run_id, str) or not run_id: raise ContractError("report run id is invalid") @@ -493,6 +504,21 @@ def build_terminal_reports( raise ContractError("report completion timestamp is invalid") if error is not None and not isinstance(error, dict): raise ContractError("report error must be an object or null") + lead_mode = snapshot["task"]["lead"]["mode"] + lead_evidence = [] if headless_leads is None else headless_leads + if not isinstance(lead_evidence, list): + raise ContractError("headless lead report evidence must be an array") + if lead_mode == "headless": + if len(lead_evidence) != len(history): + raise ContractError("headless lead report history is incomplete") + for evidence, history_entry in zip(lead_evidence, history): + if (not isinstance(evidence, dict) + or evidence.get("handoff_id") != history_entry["handoff_id"] + or evidence.get("choice", {}).get("disposition") + != history_entry["decision"]["disposition"]): + raise ContractError("headless lead report evidence changes its decision") + elif lead_evidence: + raise ContractError("host-led report cannot contain headless lead evidence") projected = [_artifact_projection(artifact) for artifact in run_artifacts] artifact_by_id = {artifact["id"]: artifact for artifact in projected} @@ -519,7 +545,11 @@ def build_terminal_reports( "Reviewer output came from the explicit offline fixture; " "it is not live-provider evidence." ) - native_counts = [attempt["native_model_requests"] for attempt in attempts] + lead_attempts = [evidence["attempt"] for evidence in lead_evidence] + all_attempts = attempts + lead_attempts + all_native_counts = [ + attempt["native_model_requests"] for attempt in all_attempts + ] receipt = { "schema_version": 1, "run_id": run_id, @@ -546,29 +576,32 @@ def build_terminal_reports( "maximum": snapshot["task"]["budget"]["max_revisions"], }, "lead": { - "mode": "host", + "mode": lead_mode, "disposition": final_decision["disposition"], "reason": final_decision["reason"], "submission_id": final_decision["submission_id"], "submission_hash": final_decision["submission_hash"], "evidence_refs": final_decision["evidence_refs"], - "usage": { - "input_tokens": None, - "output_tokens": None, - "total_tokens": None, - "source": "unavailable", - }, + "attempts": lead_attempts, + "usage": ( + lead_attempts[-1]["usage"] if lead_attempts else { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + } + ), }, "accounting": { "worker_invocations": sum( - attempt["worker_invocations"] for attempt in attempts + attempt["worker_invocations"] for attempt in all_attempts ), "native_model_requests": ( - None if any(value is None for value in native_counts) - else sum(native_counts) + None if any(value is None for value in all_native_counts) + else sum(all_native_counts) ), - "attempt_usage": [attempt["usage"] for attempt in attempts], - "host_usage_measured": False, + "attempt_usage": [attempt["usage"] for attempt in all_attempts], + "host_usage_measured": False if lead_mode == "host" else None, }, "artifacts": projected, "evidence_artifacts": evidence_projected, diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 1a755b0..45e0067 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -13,9 +13,10 @@ import threading from typing import Any +from .codex_lead_worker import freeze_codex_lead from .codex_review_worker import freeze_codex_reviewer from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported -from .reports import build_terminal_reports +from .reports import build_early_terminal_reports, build_terminal_reports from .router import load_routing from .store import ( ConflictError, @@ -24,10 +25,12 @@ Store, TERMINAL_STATES, canonical_json, + request_hash, ) from .validation import validate_task from .workflows import ( apply_lead_disposition, + decode_headless_lead_evidence, review_mode, validate_branch_review_handoff, validate_handoff_decision_evidence, @@ -53,6 +56,38 @@ def __init__(self, runtime: Path): def _store(self) -> Store: return Store(self.database, self.artifacts) + def _preparation_failure_artifacts( + self, + store: Store, + run_id: str, + task: dict[str, Any], + snapshot: dict[str, Any] | None, + error: dict[str, Any], + ) -> list[dict[str, Any]] | None: + if task.get("workflow") != "branch-review": + return None + reports = build_early_terminal_reports( + run_id=run_id, + state="failed", + task=task, + snapshot=snapshot, + run_artifacts=[], + events=store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + phase="preparing", + error=error, + ) + prepared = [] + for name in sorted(reports): + path, digest, size = store.finalize_artifact(run_id, name, reports[name]) + prepared.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + return prepared + def _freeze_package(self) -> tuple[Path, str]: source = Path(__file__).resolve().parent digest = hashlib.sha256() @@ -173,6 +208,7 @@ def _resolve_snapshot( project_id: str | None = None, run_id: str | None = None, internal_review_fixture: dict[str, Any] | None = None, + internal_lead_fixture: dict[str, Any] | None = None, ) -> dict[str, Any]: repo = resolved_repo or Path(task["project"]["repo_path"]).resolve(strict=True) base_oid = resolve_commit(repo, task["project"]["base_ref"]) @@ -250,6 +286,17 @@ def _resolve_snapshot( snapshot["internal_review_fixture"] = validate_review_document( fixture_document, task, snapshot["workspace"], ) + if internal_lead_fixture is not None: + if (task["lead"]["mode"] != "headless" + or not isinstance(internal_lead_fixture, dict) + or set(internal_lead_fixture) != {"disposition", "reason"} + or internal_lead_fixture["disposition"] + not in {"accept", "revise", "reject"} + or not isinstance(internal_lead_fixture["reason"], str)): + raise ContractError("internal lead fixture is invalid") + snapshot["internal_lead_fixture"] = json.loads( + canonical_json(internal_lead_fixture) + ) return snapshot def _continue_preparation( @@ -266,9 +313,15 @@ def _continue_preparation( task = submitted["task"] internal_delay = submitted.get("_internal_fake_delay") internal_review_fixture = submitted.get("_internal_review_fixture") + internal_lead_fixture = submitted.get("_internal_lead_fixture") store.validate_predecessor(run_id, fencing_token, supersedes_run_id) validated_supersedes_run_id = supersedes_run_id validate_task(task, require_existing_repo=True) + if (task["lead"]["mode"] == "headless" + and task["budget"]["max_worker_invocations"] < 2): + raise ContractError( + "headless branch review requires at least two worker invocations" + ) worktree = store.preparation_worktree( run_id, fencing_token, Path(task["project"]["repo_path"]), ) @@ -280,11 +333,17 @@ def _continue_preparation( project_id=project_id, run_id=run_id, internal_review_fixture=internal_review_fixture, + internal_lead_fixture=internal_lead_fixture, ) if internal_delay is None and internal_review_fixture is None: snapshot["review_adapter"] = freeze_codex_reviewer( snapshot["routing"]["roles"]["reviewer"]["selected"], ) + if (internal_delay is None and task["lead"]["mode"] == "headless" + and internal_lead_fixture is None): + snapshot["lead_adapter"] = freeze_codex_lead( + snapshot["routing"]["roles"]["lead"]["selected"], + ) package, digest = self._freeze_package() version = store.complete_preparation( run_id, @@ -298,17 +357,24 @@ def _continue_preparation( return (version, package, digest), None except (CapabilityUnavailable, ProfileUnsupported) as exc: error = {"error": exc.code, "message": str(exc)} + terminal_artifacts = self._preparation_failure_artifacts( + store, run_id, task, snapshot, error, + ) store.fail_preparation( run_id, fencing_token, error, mutable_snapshot=snapshot, supersedes_run_id=validated_supersedes_run_id, + terminal_artifacts=terminal_artifacts, ) return None, error except Exception as exc: error = {"error": "PREPARATION_FAILED", "message": str(exc)} try: + terminal_artifacts = self._preparation_failure_artifacts( + store, run_id, task, snapshot, error, + ) # Preserve validated lineage through unrelated failures; a # rejected predecessor remains only in submitted_request. store.fail_preparation( @@ -317,6 +383,7 @@ def _continue_preparation( error, mutable_snapshot=snapshot, supersedes_run_id=validated_supersedes_run_id, + terminal_artifacts=terminal_artifacts, ) except ConflictError: # Cancellation or another recovery owner may have fenced us. @@ -331,15 +398,20 @@ def start( *, _internal_fake_delay: float | None = None, _internal_review_fixture: dict[str, Any] | None = None, + _internal_lead_fixture: dict[str, Any] | None = None, ) -> dict[str, Any]: validate_task(task, require_existing_repo=True) if _internal_fake_delay is not None and _internal_review_fixture is not None: raise ContractError("internal lifecycle fixtures are mutually exclusive") + if _internal_lead_fixture is not None and _internal_fake_delay is not None: + raise ContractError("internal lifecycle fixtures are mutually exclusive") submitted = {"task": task, "supersedes_run_id": supersedes_run_id} if _internal_fake_delay is not None: submitted["_internal_fake_delay"] = _internal_fake_delay if _internal_review_fixture is not None: submitted["_internal_review_fixture"] = _internal_review_fixture + if _internal_lead_fixture is not None: + submitted["_internal_lead_fixture"] = _internal_lead_fixture store = self._store() try: claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}") @@ -377,7 +449,12 @@ def status(self, run_id: str) -> dict[str, Any]: if run["state"] == "blocked": next_action = "recovery_file_required" elif run["state"] == "awaiting_host" and run["phase"] is None: - next_action = "claim_handoff" + try: + snapshot = self._review_snapshot(run) + headless = snapshot["task"]["lead"]["mode"] == "headless" + except (ConflictError, KeyError, TypeError): + headless = False + next_action = "continue_headless_lead" if headless else "claim_handoff" elif run["state"] == "awaiting_host": next_action = "handoff_submission_saved" else: @@ -450,6 +527,15 @@ def handoff_claim( decoded = self._decode_claim(prior_claim) if prior_claim is not None else None store = self._store() try: + run = store.run(run_id) + handoff_before_claim = store.handoff_snapshot(run_id) + if (handoff_before_claim is not None + and handoff_before_claim.packet.get("workflow") == "branch-review" + and self._review_snapshot(run)["task"]["lead"]["mode"] + == "headless"): + raise ConflictError( + "headless branch review does not accept a host claim" + ) claim = store.claim_handoff(run_id, expected_version, owner, decoded) snapshot = store.handoff_snapshot(run_id) if snapshot is None: # Defensive: claim_handoff just verified it. @@ -528,6 +614,29 @@ def _terminalize_branch_review( or hashlib.sha256(path.read_bytes()).hexdigest() != artifact["sha256"]): raise ConflictError("branch review artifact is missing or corrupt") + headless_leads = [] + if snapshot["task"]["lead"]["mode"] == "headless": + artifacts_by_name = { + artifact["name"]: artifact for artifact in run_artifacts + } + for item in history: + submission_id = item["decision"]["submission_id"] + if not submission_id.startswith("headless-"): + raise ConflictError("headless lead submission identity is invalid") + attempt_id = submission_id[len("headless-"):] + artifact = artifacts_by_name.get( + f"lead-attempt-{attempt_id}.json" + ) + if artifact is None: + raise ConflictError("headless lead evidence artifact is missing") + frozen_handoff = { + "handoff_id": item["handoff_id"], + "packet": item["packet"], + "packet_sha256": item["packet_sha256"], + } + headless_leads.append(decode_headless_lead_evidence( + Path(artifact["path"]).read_bytes(), snapshot, frozen_handoff, + )) reports = build_terminal_reports( run_id=run_id, state=terminal_state, @@ -537,6 +646,7 @@ def _terminalize_branch_review( events=store.events_for_run(run_id), completed_at=datetime.now(timezone.utc).isoformat(), error=error, + headless_leads=headless_leads, ) prepared = [] for name in sorted(reports): @@ -651,6 +761,111 @@ def _continue_branch_review_submission( error, ) + def _saved_headless_lead( + self, + store: Store, + run_id: str, + snapshot: dict[str, Any], + handoff: HandoffSnapshot, + ) -> tuple[dict[str, Any], dict[str, Any]] | None: + frozen_handoff = { + "handoff_id": handoff.handoff_id, + "packet": handoff.packet, + "packet_sha256": handoff.packet_sha256, + } + for attempt in reversed(store.attempts_for_run(run_id)): + if attempt.get("role") != "lead" or attempt.get("status") != "finished": + continue + artifact = store.artifact_named( + run_id, f"lead-attempt-{attempt['id']}.json", + ) + if artifact is None: + continue + path = Path(artifact["path"]) + try: + content = path.read_bytes() + except OSError as exc: + raise ConflictError("saved headless lead evidence is missing") from exc + if (len(content) != artifact["byte_size"] + or hashlib.sha256(content).hexdigest() != artifact["sha256"]): + raise ConflictError("saved headless lead evidence is corrupt") + try: + evidence = decode_headless_lead_evidence( + content, snapshot, frozen_handoff, + ) + except ContractError: + continue + return attempt, evidence + return None + + def _continue_headless_lead( + self, + store: Store, + run_id: str, + run: dict[str, Any], + handoff: HandoffSnapshot, + snapshot: dict[str, Any], + ) -> dict[str, Any]: + saved = self._saved_headless_lead( + store, run_id, snapshot, handoff, + ) + if saved is None: + queued = store.queue_headless_lead(run_id, run["version"]) + if queued["action"] != "queued": + raise ConflictError( + "headless lead worker invocation budget is exhausted" + ) + prepared = store.run(run_id) + package, digest = self._verified_package(prepared) + return { + "action": "headless_lead_queued", + "state": prepared["state"], + "version": prepared["version"], + "replayed_continuation": False, + "launch": (prepared["version"], package, digest), + } + + attempt, evidence = saved + claim = store.claim_handoff( + run_id, + run["version"], + f"headless-lead:{attempt['id']}", + ) + choice = evidence["choice"] + body = { + "schema_version": 1, + "submission_id": f"headless-{attempt['id']}", + "disposition": choice["disposition"], + "reason": choice["reason"], + "evidence_refs": [ + { + "artifact_id": reference["artifact_id"], + "sha256": reference["sha256"], + } + for reference in handoff.packet["artifacts"] + ], + } + decision = {**body, "submission_hash": request_hash(body)} + self._review_gate(store, run_id, handoff, snapshot, decision) + submission = store.record_handoff_submission(run_id, claim, decision) + entry = store.recorded_handoff_submission(run_id, handoff.handoff_id) + if entry is None: + raise ConflictError("recorded headless lead submission is missing") + continuation = self._continue_branch_review_submission( + store, run_id, handoff, snapshot, entry, + ) + launch = continuation.pop("launch") + current = store.run(run_id) + return { + "action": continuation["action"], + "state": current["state"], + "version": current["version"], + "submission_id": submission.submission_id, + "disposition": submission.disposition, + "replayed_continuation": continuation["replayed_continuation"], + "launch": launch, + } + def handoff_complete( self, run_id: str, @@ -667,6 +882,11 @@ def handoff_complete( handoff = store.handoff_snapshot_by_id(run_id, decoded.handoff_id) branch_review = handoff.packet.get("workflow") == "branch-review" snapshot = self._review_snapshot(run) if branch_review else None + if (branch_review and snapshot["task"]["lead"]["mode"] == "headless" + and not decoded.owner_id.startswith("headless-lead:")): + raise ConflictError( + "headless branch review does not accept a host completion" + ) if branch_review: self._review_gate(store, run_id, handoff, snapshot, decision) submission = store.record_handoff_submission(run_id, decoded, decision) @@ -712,7 +932,27 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st try: run = store.run(run_id) if run["state"] in TERMINAL_STATES: raise ConflictError("terminal run cannot resume; start a superseding run") - if run["state"] == "awaiting_host" and run["phase"] == "handoff_submitted": + if run["state"] == "awaiting_host" and run["phase"] is None: + handoff = store.handoff_snapshot(run_id) + if handoff is None or handoff.packet.get("workflow") != "branch-review": + raise ConflictError("run has no resumable branch review handoff") + snapshot = self._review_snapshot(run) + if snapshot["task"]["lead"]["mode"] == "headless": + continuation = self._continue_headless_lead( + store, run_id, run, handoff, snapshot, + ) + launch = continuation.pop("launch") + current = store.run(run_id) + branch_response = { + "run_id": run_id, + "disposition": continuation["action"], + "state": current["state"], + "version": current["version"], + "launched": launch is not None, + } + else: + raise ConflictError("host-led handoff must be completed by its host") + elif run["state"] == "awaiting_host" and run["phase"] == "handoff_submitted": handoff = store.handoff_snapshot(run_id) if handoff is None or handoff.packet.get("workflow") != "branch-review": raise ConflictError("run has no resumable branch review submission") diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index d1adb7a..28bbcb2 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import ContractError -SUPPORTED_SCHEMA_VERSION = 5 +SUPPORTED_SCHEMA_VERSION = 6 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -425,8 +425,15 @@ def fail_preparation( *, mutable_snapshot: Any | None = None, supersedes_run_id: str | None = None, + terminal_artifacts: list[dict[str, Any]] | None = None, ) -> int: encoded = canonical_json(error) + prepared = ( + self._prepare_exact_artifacts( + run_id, terminal_artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS, + ) + if terminal_artifacts is not None else None + ) self.connection.execute("BEGIN IMMEDIATE") try: row = self.connection.execute( @@ -443,10 +450,18 @@ def fail_preparation( if (not predecessor or predecessor["project_id"] != row["project_id"] or predecessor["state"] not in TERMINAL_STATES): raise ConflictError("superseded run must be terminal and belong to the same project") - path, digest, size, now = self._terminal_receipt(run_id, "failed", "preparing", error) - version = self._reference_terminal_receipt( - run_id, row["version"], path, digest, size, now, - ) + now = _utc_now() + if prepared is None: + path, digest, size, now = self._terminal_receipt( + run_id, "failed", "preparing", error, now=now, + ) + version = self._reference_terminal_receipt( + run_id, row["version"], path, digest, size, now, + ) + else: + version, _ = self._reference_prepared_artifacts( + run_id, row["version"], prepared, + ) version += 1 snapshot = canonical_json(mutable_snapshot) if mutable_snapshot is not None else None self.connection.execute( @@ -581,9 +596,18 @@ def store_artifact(self, run_id: str, name: str, content: bytes) -> str: path, digest, _ = self.finalize_artifact(run_id, name, content) return self.reference_artifact(run_id, name, path, digest) - def reserve_attempt(self, run_id: str, expected_version: int, owner_id: str, package_digest: str) -> AttemptReservation: + def reserve_attempt( + self, + run_id: str, + expected_version: int, + owner_id: str, + package_digest: str, + role: str = "worker", + ) -> AttemptReservation: if not owner_id or not package_digest: raise ContractError("supervisor owner and package digest are required") + if role not in {"worker", "implementer", "reviewer", "lead", "researcher"}: + raise ContractError("attempt role is invalid") self.connection.execute("BEGIN IMMEDIATE") try: run = self.connection.execute("SELECT project_id,worktree_path,state,phase,version FROM runs WHERE id=?", (run_id,)).fetchone() @@ -598,7 +622,7 @@ def reserve_attempt(self, run_id: str, expected_version: int, owner_id: str, pac supervisor_token, attempt_id, attempt_token = old + 1, str(uuid.uuid4()), uuid.uuid4().hex now, version = _utc_now(), expected_version + 1 self.connection.execute("INSERT OR REPLACE INTO supervisor_claims(run_id,owner_id,fencing_token,package_digest,heartbeat_at,active) VALUES(?,?,?,?,?,1)", (run_id, owner_id, supervisor_token, package_digest, now)) - self.connection.execute("INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token,status,heartbeat_at,package_digest,created_at) VALUES(?,?,?,?,?,'reserved',?,?,?)", (attempt_id, run_id, run["project_id"], run["worktree_path"], attempt_token, now, package_digest, now)) + self.connection.execute("INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token,status,heartbeat_at,package_digest,created_at,role) VALUES(?,?,?,?,?,'reserved',?,?,?,?)", (attempt_id, run_id, run["project_id"], run["worktree_path"], attempt_token, now, package_digest, now, role)) self.connection.execute("UPDATE runs SET phase='launching',version=?,updated_at=? WHERE id=?", (version, now, run_id)) event = canonical_json({"attempt_id": attempt_id, "supervisor_token": supervisor_token}) self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'supervisor.claimed',?,?)", (run_id, version, event, now)) @@ -997,6 +1021,32 @@ def commit_durable_handoff( (run_id,), ).fetchone()[0] handoff_id, now = str(uuid.uuid4()), _utc_now() + from .reports import build_handoff_reports, handoff_report_names + handoff_reports = build_handoff_reports( + run_id=run_id, + handoff_id=handoff_id, + sequence=sequence, + packet=frozen_packet, + packet_sha256=packet_sha256, + created_at=now, + ) + report_artifacts = [] + for name, content in handoff_reports.items(): + path, digest, size = self.finalize_artifact(run_id, name, content) + report_artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + prepared_reports = self._prepare_exact_artifacts( + run_id, + report_artifacts, + frozenset(handoff_report_names(sequence)), + ) + version, _ = self._reference_prepared_artifacts( + run_id, version, prepared_reports, + ) version += 1 self.connection.execute( "INSERT INTO handoffs(id,run_id,sequence,packet_json,packet_sha256,status," @@ -1031,6 +1081,162 @@ def commit_durable_handoff( self.connection.execute("ROLLBACK") raise + def queue_headless_lead( + self, + run_id: str, + expected_version: int, + ) -> dict[str, Any]: + """Pause an open handoff only long enough to launch its configured lead.""" + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT state,phase,version,mutable_snapshot FROM runs WHERE id=?", + (run_id,), + ).fetchone() + handoff = self.connection.execute( + "SELECT id,status,sequence FROM handoffs WHERE run_id=? " + "ORDER BY sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + if not row or not handoff: + raise ConflictError("headless lead handoff is missing") + if row["version"] != expected_version: + raise ConflictError("headless lead run version changed") + try: + snapshot = json.loads(row["mutable_snapshot"]) + budget = snapshot["task"]["budget"]["max_worker_invocations"] + lead_mode = snapshot["task"]["lead"]["mode"] + except (TypeError, KeyError, json.JSONDecodeError) as exc: + raise ConflictError("frozen headless lead configuration is invalid") from exc + if lead_mode != "headless" or type(budget) is not int or budget < 1: + raise ConflictError("run is not configured for a headless lead") + if (row["state"] != "awaiting_host" or row["phase"] is not None + or handoff["status"] != "open"): + raise ConflictError("headless lead handoff is not queueable") + active = self.connection.execute( + "SELECT 1 FROM supervisor_claims WHERE run_id=? AND active=1", + (run_id,), + ).fetchone() + if active: + raise ConflictError("headless lead already has an active supervisor") + invocations = self.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()[0] + if invocations >= budget: + self.connection.execute("COMMIT") + return { + "action": "budget_exhausted", + "version": row["version"], + "worker_invocations": invocations, + } + now, version = _utc_now(), row["version"] + 1 + self.connection.execute( + "UPDATE runs SET state='queued',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff["id"], + "sequence": handoff["sequence"], + "worker_invocations": invocations, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.headless_lead_queued',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return { + "action": "queued", + "version": version, + "worker_invocations": invocations, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def commit_headless_lead( + self, + run_id: str, + attempt_token: str, + artifacts: list[dict[str, Any]], + metadata: Any, + ) -> str: + """Import one valid headless-lead result and reopen its frozen handoff.""" + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=False, + ) + encoded_metadata = canonical_json(metadata) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,role,stdout_artifact_id,stderr_artifact_id," + "output_metadata FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + handoff = self.connection.execute( + "SELECT id,status,sequence FROM handoffs WHERE run_id=? " + "ORDER BY sequence DESC LIMIT 1", + (run_id,), + ).fetchone() + if not run or not attempt or not handoff: + raise ConflictError("headless lead import is fenced") + if (attempt["status"] == "finished" + and run["state"] == "awaiting_host" + and handoff["status"] == "open"): + self.connection.execute("COMMIT") + return "awaiting_host" + if (attempt["status"] != "running" or attempt["role"] != "lead" + or run["state"] != "running" or run["phase"] is not None + or handoff["status"] != "open"): + raise ConflictError("headless lead import is fenced") + evidence_name = f"lead-attempt-{attempt['id']}.json" + if evidence_name not in {item[0] for item in prepared}: + raise ContractError("headless lead evidence artifact is missing") + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='awaiting_host',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "handoff_id": handoff["id"], + "evidence": evidence_name, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.headless_lead_ready',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return "awaiting_host" + except Exception: + self.connection.execute("ROLLBACK") + raise + def block_recovery(self, run_id: str, attempt_token: str, reason: str, *, release_writer: bool = False) -> int: self.connection.execute("BEGIN IMMEDIATE") try: @@ -1922,6 +2128,7 @@ def requeue_review_revision( budget = snapshot["task"]["budget"] max_revisions = budget["max_revisions"] max_invocations = budget["max_worker_invocations"] + lead_mode = snapshot["task"]["lead"]["mode"] except (KeyError, TypeError) as exc: raise ConflictError("frozen review budget is missing") from exc if (type(max_revisions) is not int or max_revisions < 0 @@ -1937,7 +2144,9 @@ def requeue_review_revision( invocations = self.connection.execute( "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), ).fetchone()[0] - if revisions > max_revisions or invocations >= max_invocations: + required_invocations = 2 if lead_mode == "headless" else 1 + if (revisions > max_revisions + or invocations + required_invocations > max_invocations): self.connection.execute("COMMIT") return { "action": "budget_exhausted", @@ -2234,6 +2443,14 @@ def attempt(self, run_id: str) -> dict[str, Any] | None: row = self.connection.execute("SELECT * FROM attempts WHERE run_id=? ORDER BY created_at DESC LIMIT 1", (run_id,)).fetchone() return dict(row) if row else None + def attempts_for_run(self, run_id: str) -> list[dict[str, Any]]: + return [ + dict(row) for row in self.connection.execute( + "SELECT * FROM attempts WHERE run_id=? ORDER BY created_at,id", + (run_id,), + ) + ] + def run(self, run_id: str) -> dict[str, Any]: row = self.connection.execute("SELECT * FROM runs WHERE id=?", (run_id,)).fetchone() if not row: raise ContractError("run does not exist") diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index bed7746..1107c0f 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -20,7 +20,7 @@ from .contracts import ContractError, LaunchSpec from .reports import build_early_terminal_reports from .store import AttemptReservation, ConflictError, Store, canonical_json -from .workflows import decode_branch_review_evidence +from .workflows import decode_branch_review_evidence, decode_headless_lead_evidence def _open_stdin_artifact(path: str) -> BinaryIO: @@ -209,8 +209,19 @@ def launch(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: stdout.start(); stderr.start() return RunningAttempt(reservation, process, started, stdout, stderr) - def launch_durable(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: str, package_digest: str) -> DurableAttempt: - reservation = self.store.reserve_attempt(run_id, expected_version, owner_id, package_digest) + def launch_durable( + self, + run_id: str, + expected_version: int, + spec: LaunchSpec, + owner_id: str, + package_digest: str, + *, + role: str = "worker", + ) -> DurableAttempt: + reservation = self.store.reserve_attempt( + run_id, expected_version, owner_id, package_digest, role, + ) directory = self.store.artifacts / run_id / f".{reservation.attempt_id}.spool" directory.mkdir(parents=True, exist_ok=False) paths = {name: str(directory / filename) for name, filename in { @@ -408,8 +419,36 @@ def import_durable(self, run_id: str) -> str: workflow_review = ( "internal_review_fixture" in snapshot or "review_adapter" in snapshot ) + role = attempt.get("role", "worker") semantic_error=None - if (workflow_review and not receipt["cancelled"] + if (role == "lead" and workflow_review and not receipt["cancelled"] + and not receipt["timed_out"] and receipt["returncode"]==0): + try: + handoff = self.store.handoff_snapshot(run_id) + if handoff is None or handoff.status != "open": + raise ContractError("headless lead handoff is missing") + frozen_handoff = { + "handoff_id": handoff.handoff_id, + "packet": handoff.packet, + "packet_sha256": handoff.packet_sha256, + } + evidence = decode_headless_lead_evidence( + captures["stdout"], snapshot, frozen_handoff, + ) + content = (canonical_json(evidence) + "\n").encode() + name = f"lead-attempt-{attempt['id']}.json" + path,digest,size=self.store.finalize_artifact(run_id,name,content) + artifacts.append({ + "name":name,"path":path,"sha256":digest,"byte_size":size, + }) + return self.store.commit_headless_lead( + run_id, attempt["attempt_token"], artifacts, metadata, + ) + except ContractError as exc: + semantic_error=str(exc) + receipt["error"]="HEADLESS_LEAD_OUTPUT_INVALID" + receipt["message"]=semantic_error + elif (workflow_review and not receipt["cancelled"] and not receipt["timed_out"] and receipt["returncode"]==0): try: return self._commit_review_handoff( @@ -423,9 +462,26 @@ def import_durable(self, run_id: str) -> str: payload={"returncode":receipt["returncode"],"receipt":"result-receipt.json"} if receipt["timed_out"]: payload["error"]="TIMEOUT" if semantic_error: - payload["error"]="WORKFLOW_OUTPUT_INVALID" + payload["error"]=( + "HEADLESS_LEAD_OUTPUT_INVALID" + if role == "lead" else "WORKFLOW_OUTPUT_INVALID" + ) payload["message"]=semantic_error if workflow_review: + prior_attempts = [] + if role == "lead": + handoff = self.store.handoff_snapshot(run_id) + if handoff is not None and isinstance(handoff.packet, dict): + prior = handoff.packet.get("attempt") + if isinstance(prior, dict): + prior_attempts.append({ + "id": handoff.packet.get("attempt_id"), + "status": "succeeded", + **prior, + "review": handoff.packet.get("review"), + "checks": handoff.packet.get("checks"), + "evaluation": handoff.packet.get("evaluation"), + }) if receipt["cancelled"]: report_error = None elif receipt["timed_out"]: @@ -435,13 +491,23 @@ def import_durable(self, run_id: str) -> str: } elif semantic_error: report_error = { - "error": "WORKFLOW_OUTPUT_INVALID", + "error": ( + "HEADLESS_LEAD_OUTPUT_INVALID" + if role == "lead" else "WORKFLOW_OUTPUT_INVALID" + ), "message": semantic_error, } else: report_error = { - "error": "REVIEW_WORKER_FAILED", - "message": "branch review worker exited before producing a valid handoff", + "error": ( + "HEADLESS_LEAD_FAILED" + if role == "lead" else "REVIEW_WORKER_FAILED" + ), + "message": ( + "headless lead exited before producing a valid disposition" + if role == "lead" + else "branch review worker exited before producing a valid handoff" + ), "returncode": receipt["returncode"], } payload.update(report_error) @@ -450,17 +516,22 @@ def import_durable(self, run_id: str) -> str: state=terminal, task=snapshot["task"], snapshot=snapshot, - run_artifacts=artifacts, + run_artifacts=( + self.store.artifacts_for_run(run_id) + artifacts + if role == "lead" else artifacts + ), events=self.store.events_for_run(run_id), completed_at=datetime.now(timezone.utc).isoformat(), - phase="reviewer", + phase="lead" if role == "lead" else "reviewer", error=report_error, attempt={ "id": attempt["id"], + "role": role, "returncode": receipt["returncode"], "cancelled": receipt["cancelled"], "timed_out": receipt["timed_out"], }, + prior_attempts=prior_attempts, ) for name in sorted(reports): content = reports[name] diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index 9b3edc0..6b67c98 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -14,6 +14,7 @@ MAX_REVIEW_BYTES = 512 * 1024 +MAX_LEAD_BYTES = 128 * 1024 MAX_EVIDENCE_BYTES = 1024 * 1024 MAX_FINDINGS = 100 MAX_TEXT_CHARS = 20_000 @@ -72,6 +73,25 @@ def review_output_schema() -> dict[str, Any]: } +def lead_output_schema() -> dict[str, Any]: + """Return the strict native structured-output schema for one lead choice.""" + return { + "type": "object", + "additionalProperties": False, + "required": [ + "schema_version", "candidate_sha256", "disposition", "reason", + ], + "properties": { + "schema_version": {"type": "integer", "enum": [1]}, + "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "disposition": { + "type": "string", "enum": ["accept", "revise", "reject"], + }, + "reason": {"type": "string"}, + }, + } + + def _exact( value: Any, fields: set[str], @@ -448,6 +468,226 @@ def build_review_prompt(task: dict[str, Any], workspace: dict[str, Any]) -> str: ]) +def validate_headless_lead_choice( + value: dict[str, Any], + packet: dict[str, Any], +) -> dict[str, Any]: + """Validate a headless lead's bounded choice against trusted review gates.""" + choice = _exact(value, { + "schema_version", "candidate_sha256", "disposition", "reason", + }, "headless lead choice") + if choice["schema_version"] != 1 or type(choice["schema_version"]) is not int: + raise ContractError("headless lead choice schema_version is invalid") + if _sha256( + choice["candidate_sha256"], "headless lead candidate_sha256", + ) != packet.get("candidate_sha256"): + raise ContractError("headless lead choice targets a different candidate") + disposition = choice["disposition"] + if disposition not in {"accept", "revise", "reject"}: + raise ContractError("headless lead disposition is invalid") + reason = choice["reason"] + if not isinstance(reason, str) or len(reason) > MAX_TEXT_CHARS: + raise ContractError("headless lead reason is invalid") + if disposition in {"revise", "reject"} and not reason.strip(): + raise ContractError("headless lead revise/reject requires a reason") + if disposition == "accept" and packet.get("evaluation", {}).get("accept_allowed") is not True: + raise ContractError("headless lead acceptance is blocked by required evidence") + return json.loads(canonical_json(choice)) + + +def decode_headless_lead_choice( + payload: bytes | str, + packet: dict[str, Any], +) -> dict[str, Any]: + return validate_headless_lead_choice( + _strict_json_object(payload, "headless lead output", maximum=MAX_LEAD_BYTES), + packet, + ) + + +def build_lead_prompt(task: dict[str, Any], packet: dict[str, Any]) -> str: + """Build the single frozen evidence-disposition prompt for a headless lead.""" + validate_task(task) + if task["workflow"] != "branch-review" or task["lead"]["mode"] != "headless": + raise ContractError("headless lead prompt requires a headless branch review") + if not isinstance(packet, dict): + raise ContractError("headless lead packet must be an object") + assignment = { + "goal": task["goal"], + "acceptance": task["acceptance"], + "candidate_sha256": packet.get("candidate_sha256"), + "review": packet.get("review"), + "checks": packet.get("checks"), + "evaluation": packet.get("evaluation"), + } + return "\n".join([ + "You are the single read-only lead for one frozen branch-review handoff.", + "Do not edit files, run commands, publish, delegate, or broaden scope.", + "Choose exactly one disposition: accept, revise, or reject.", + "Acceptance is forbidden when evaluation.accept_allowed is false.", + "Return exactly one JSON object and no Markdown or surrounding prose.", + "The object must contain exactly schema_version, candidate_sha256, disposition, reason.", + "Frozen handoff evidence:", + canonical_json(assignment), + ]) + + +def validate_headless_lead_evidence( + value: dict[str, Any], + snapshot: dict[str, Any], + handoff: dict[str, Any], +) -> dict[str, Any]: + """Validate one provider-bound headless lead attempt and its exact handoff.""" + document = _exact(value, { + "schema_version", "workflow", "candidate_sha256", "handoff_id", + "packet_sha256", "choice", "attempt", + }, "headless lead evidence") + if document["schema_version"] != 1 or type(document["schema_version"]) is not int: + raise ContractError("headless lead evidence schema_version is invalid") + if document["workflow"] != "branch-review": + raise ContractError("headless lead evidence workflow is invalid") + if not isinstance(snapshot, dict) or not isinstance(handoff, dict): + raise ContractError("headless lead frozen inputs are invalid") + task = snapshot.get("task") + packet = handoff.get("packet") + if not isinstance(task, dict) or not isinstance(packet, dict): + raise ContractError("headless lead frozen inputs are incomplete") + if task.get("lead", {}).get("mode") != "headless": + raise ContractError("headless lead evidence requires headless mode") + handoff_id = _text(document["handoff_id"], "headless lead handoff id", maximum=200) + if handoff_id != handoff.get("handoff_id"): + raise ContractError("headless lead evidence targets a different handoff") + packet_sha256 = hashlib.sha256(canonical_json(packet).encode()).hexdigest() + if _sha256(document["packet_sha256"], "headless lead packet sha256") != packet_sha256: + raise ContractError("headless lead evidence changes the handoff packet") + if handoff.get("packet_sha256") != packet_sha256: + raise ContractError("headless lead input packet hash is invalid") + choice = validate_headless_lead_choice(document["choice"], packet) + if _sha256( + document["candidate_sha256"], "headless lead evidence candidate_sha256", + ) != choice["candidate_sha256"]: + raise ContractError("headless lead evidence changes the candidate") + + attempt = _exact(document["attempt"], { + "role", "selected_profile", "prompt_sha256", "observed_identity", + "native_ids", "worker_invocations", "native_model_requests", "usage", + }, "headless lead attempt evidence") + if attempt["role"] != "lead": + raise ContractError("headless lead attempt role is invalid") + try: + frozen_lead = snapshot["routing"]["roles"]["lead"]["selected"] + except (KeyError, TypeError) as exc: + raise ContractError("frozen lead selection is missing") from exc + if canonical_json(attempt["selected_profile"]) != canonical_json(frozen_lead): + raise ContractError("headless lead attempt changes the selected profile") + prompt_sha256 = hashlib.sha256(build_lead_prompt(task, packet).encode()).hexdigest() + if _sha256(attempt["prompt_sha256"], "headless lead prompt sha256") != prompt_sha256: + raise ContractError("headless lead prompt hash does not match the handoff") + + adapter = snapshot.get("lead_adapter") + observed = attempt["observed_identity"] + native_ids = attempt["native_ids"] + if adapter is None: + if observed is not None or native_ids != {}: + raise ContractError("fixture lead cannot claim a native observed identity") + else: + if not isinstance(adapter, dict): + raise ContractError("frozen lead adapter is invalid") + identity = _exact(observed, { + "harness", "harness_version", "model_provider", "model_id", "effort", + "permission_policy", "verification", + }, "observed lead identity") + expected_identity = { + "harness": adapter.get("harness"), + "harness_version": adapter.get("harness_version"), + "model_provider": adapter.get("model_provider"), + "model_id": frozen_lead["profile"]["model_id"], + "effort": frozen_lead["profile"]["effort"]["value"], + "permission_policy": frozen_lead["profile"]["permission_policy"], + "verification": "verified", + } + if canonical_json(identity) != canonical_json(expected_identity): + raise ContractError("observed lead identity does not match the frozen adapter") + ids = _exact(native_ids, {"thread_id", "turn_id"}, "native lead ids") + for field in ("thread_id", "turn_id"): + _text(ids[field], f"native lead {field}", maximum=500) + if attempt["worker_invocations"] != 1 or type(attempt["worker_invocations"]) is not int: + raise ContractError("headless lead worker invocation accounting is invalid") + native_requests = attempt["native_model_requests"] + if native_requests is not None and ( + type(native_requests) is not int or native_requests < 0): + raise ContractError("headless lead native request count is invalid") + usage = _exact(attempt["usage"], { + "input_tokens", "output_tokens", "total_tokens", "source", + }, "headless lead usage") + for field in ("input_tokens", "output_tokens", "total_tokens"): + if usage[field] is not None and ( + type(usage[field]) is not int or usage[field] < 0): + raise ContractError("headless lead token usage is invalid") + if usage["source"] not in {"native_reported", "unavailable"}: + raise ContractError("headless lead usage source is invalid") + if usage["source"] == "unavailable" and any( + usage[field] is not None + for field in ("input_tokens", "output_tokens", "total_tokens") + ): + raise ContractError("unavailable lead usage cannot invent token counts") + return json.loads(canonical_json(document)) + + +def make_headless_lead_evidence( + snapshot: dict[str, Any], + handoff: dict[str, Any], + choice: dict[str, Any], + *, + observed_identity: dict[str, Any] | None = None, + native_ids: dict[str, str] | None = None, + native_model_requests: int | None = None, + usage: dict[str, Any] | None = None, +) -> dict[str, Any]: + packet = handoff["packet"] + normalized_choice = validate_headless_lead_choice(choice, packet) + document = { + "schema_version": 1, + "workflow": "branch-review", + "candidate_sha256": normalized_choice["candidate_sha256"], + "handoff_id": handoff["handoff_id"], + "packet_sha256": handoff["packet_sha256"], + "choice": normalized_choice, + "attempt": { + "role": "lead", + "selected_profile": snapshot["routing"]["roles"]["lead"]["selected"], + "prompt_sha256": hashlib.sha256( + build_lead_prompt(snapshot["task"], packet).encode() + ).hexdigest(), + "observed_identity": observed_identity, + "native_ids": native_ids if native_ids is not None else {}, + "worker_invocations": 1, + "native_model_requests": native_model_requests, + "usage": usage if usage is not None else { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + }, + } + return validate_headless_lead_evidence(document, snapshot, handoff) + + +def decode_headless_lead_evidence( + payload: bytes | str, + snapshot: dict[str, Any], + handoff: dict[str, Any], +) -> dict[str, Any]: + return validate_headless_lead_evidence( + _strict_json_object( + payload, "headless lead evidence", maximum=MAX_EVIDENCE_BYTES, + ), + snapshot, + handoff, + ) + + def validate_branch_review_evidence( value: dict[str, Any], snapshot: dict[str, Any], diff --git a/test/core/fakes/codex_review_cli.py b/test/core/fakes/codex_review_cli.py index 6d75d41..db5bcf3 100755 --- a/test/core/fakes/codex_review_cli.py +++ b/test/core/fakes/codex_review_cli.py @@ -86,16 +86,24 @@ continue prompt = params["input"][0]["text"] assignment = json.loads(prompt.splitlines()[-1]) - review = { - "schema_version": 1, - "candidate_sha256": assignment["candidate_sha256"], - "base_oid": assignment["base_oid"], - "target_oid": assignment["target_oid"], - "review_mode": assignment["review_mode"], - "verdict": "clean", - "summary": "The bounded native fixture found no supported defect.", - "findings": [], - } + if "single read-only lead" in prompt: + review = { + "schema_version": 1, + "candidate_sha256": assignment["candidate_sha256"], + "disposition": "reject" if model.endswith("lead-reject") else "accept", + "reason": "The frozen review and checks support this disposition.", + } + else: + review = { + "schema_version": 1, + "candidate_sha256": assignment["candidate_sha256"], + "base_oid": assignment["base_oid"], + "target_oid": assignment["target_oid"], + "review_mode": assignment["review_mode"], + "verdict": "clean", + "summary": "The bounded native fixture found no supported defect.", + "findings": [], + } output = ( "{}" if model.endswith("malformed") else json.dumps(review, sort_keys=True, separators=(",", ":")) diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 78e1ad3..20d1681 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -291,7 +291,7 @@ def build_python(): return candidate return None - def test_installed_wheel_contains_and_applies_migrations_through_five(self): + def test_installed_wheel_contains_and_applies_migrations_through_six(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") @@ -333,7 +333,9 @@ def test_installed_wheel_contains_and_applies_migrations_through_five(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 5 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 6 + attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} + assert "role" in attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} assert {"package_path", "package_digest", "supersedes_run_id"} <= columns assert store.connection.execute( diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 20b4a5e..2ac45ac 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 5, + 6, ) tables = { row[0] @@ -618,7 +618,7 @@ def build_python(): return candidate return None - def test_installed_wheel_applies_schema_four_to_five(self): + def test_installed_wheel_applies_schema_four_to_six(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel") @@ -684,7 +684,7 @@ def test_installed_wheel_applies_schema_four_to_five(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 5 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 6 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_m2_cross_process.py b/test/core/test_m2_cross_process.py index 9c3c8ea..cb27764 100644 --- a/test/core/test_m2_cross_process.py +++ b/test/core/test_m2_cross_process.py @@ -16,6 +16,7 @@ sys.path.insert(0, str(ROOT / "plugin" / "core" / "src")) from devsquad.service import Service +from devsquad.reports import TERMINAL_REPORT_NAMES from devsquad.store import Store, request_hash from devsquad.supervisor import inspect_process from devsquad_test_fixtures import branch_review_routing_documents @@ -164,7 +165,10 @@ def test_identical_public_starts_share_one_run_across_processes(self): run_id = outcomes[0][1] result = Service(self.runtime).result(run_id) self.assertTrue(result["ready"]) - self.assertEqual([item["name"] for item in result["artifacts"]], ["result-receipt.json"]) + self.assertEqual( + {item["name"] for item in result["artifacts"]}, + set(TERMINAL_REPORT_NAMES), + ) def test_changed_body_conflicts_with_same_key_across_processes(self): changed = json.loads(json.dumps(self.task)) diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index ff04b8c..07ae352 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -135,6 +135,35 @@ def start_waiting(self, key): self.assertEqual(waiting["state"], "awaiting_host") return started["run_id"], waiting + def configure_fixture_headless(self): + profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) + lead = dict(profiles["profiles"][0]) + lead.update({ + "id": "fixture-lead", + "model_family": "fixture-family-lead", + "model_id": "fixture-lead-model", + }) + profiles["profiles"].append(lead) + profiles["bindings"]["lead.primary"] = { + "profile_id": "fixture-lead", "version": 1, + } + policy = json.loads((self.repo / "devsquad/policy.json").read_text()) + policy["roles"]["lead"] = [{"kind": "alias", "id": "lead.primary"}] + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "headless routing"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["lead"] = {"mode": "headless"} + self.task["budget"]["max_worker_invocations"] = 2 + def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self): (self.repo / "notes.txt").write_text("unrelated local work\n") before_head = self.git_bytes("rev-parse", "HEAD") @@ -177,6 +206,8 @@ def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self) f"checks-{attempt_id}.json", f"evaluation-{attempt_id}.json", f"review-attempt-{attempt_id}.json", + "handoff.json", + "handoff.md", } <= names) self.assertNotIn("result-receipt.json", names) handoff = store.handoff_snapshot(started["run_id"]) @@ -551,6 +582,93 @@ def test_native_codex_faults_never_become_valid_reviews(self): self.assertEqual(hashlib.sha256(saved).hexdigest(), entry["sha256"]) self.assertEqual(len(saved), entry["byte_size"]) + def test_native_codex_headless_lead_verifies_its_own_identity_and_usage(self): + profiles = { + "schema_version": 1, + "profiles": [ + { + "id": profile_id, + "harness": "codex", + "model_family": "gpt-fixture", + "model_id": model_id, + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "codex-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["native-fixture"], + } + for profile_id, model_id in ( + ("native-reviewer", "gpt-fake-review"), + ("native-lead", "gpt-fake-lead"), + ) + ], + "bindings": {}, + } + policy = { + "schema_version": 1, + "id": "native-headless-policy", + "version": 1, + "roles": { + "reviewer": [{"kind": "profile", "id": "native-reviewer"}], + "lead": [{"kind": "profile", "id": "native-lead"}], + }, + "task_classes": {"fixture-review-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "codex-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "native headless config"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["lead"] = {"mode": "headless"} + self.task["budget"]["max_worker_invocations"] = 2 + fake_bin = self.root / "native-headless-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to(ROOT / "test/core/fakes/codex_review_cli.py") + fake_home = self.root / "native-headless-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) + with patch.dict(os.environ, { + "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + "CODEX_HOME": str(fake_home), + }, clear=False): + started = self.service.start(self.task, "native-headless") + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "succeeded") + result = self.service.result(started["run_id"]) + receipt_artifact = next( + artifact for artifact in result["artifacts"] + if artifact["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["lead"]["mode"], "headless") + self.assertEqual( + receipt["lead"]["attempts"][0]["observed_identity"]["model_id"], + "gpt-fake-lead", + ) + self.assertEqual(receipt["lead"]["usage"]["total_tokens"], 160) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(receipt["accounting"]["attempt_usage"][1]["total_tokens"], 160) + def test_check_worker_failure_before_handoff_gets_full_terminal_reports(self): self.task["checks"][0]["cwd"] = "missing-check-directory" started = self.service.start( @@ -593,6 +711,89 @@ def test_waiting_handoff_report_builder_binds_packet_and_hash(self): self.assertEqual(document["packet_sha256"], handoff.packet_sha256) self.assertIn(b"claim this saved handoff", reports["handoff.md"]) + def test_headless_lead_is_a_second_fenced_attempt_and_terminalizes_automatically(self): + self.configure_fixture_headless() + started = self.service.start( + self.task, + "headless-accept", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "accept", + "reason": "The frozen review evidence is sufficient.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "succeeded") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + attempts = store.attempts_for_run(started["run_id"]) + finally: + store.close() + self.assertEqual([attempt["role"] for attempt in attempts], ["reviewer", "lead"]) + self.assertTrue(all(attempt["status"] == "finished" for attempt in attempts)) + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertIn("handoff.json", artifacts) + self.assertIn("handoff.md", artifacts) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["lead"]["mode"], "headless") + self.assertEqual(receipt["lead"]["disposition"], "accept") + self.assertEqual(len(receipt["lead"]["attempts"]), 1) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertIsNone(receipt["accounting"]["host_usage_measured"]) + + def test_invalid_headless_accept_is_a_failed_attempt_not_invented_success(self): + self.configure_fixture_headless() + self.task["checks"][0]["required_to_pass"] = True + started = self.service.start( + self.task, + "headless-invalid-accept", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "accept", + "reason": "Attempt to override a required check.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "failed") + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["phase"], "lead") + self.assertEqual(receipt["lead"]["status"], "failed") + self.assertEqual(receipt["error"]["error"], "HEADLESS_LEAD_FAILED") + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual( + [attempt["role"] for attempt in receipt["attempts"]], + ["reviewer", "lead"], + ) + + def test_headless_revise_repeats_review_and_stops_at_both_budgets(self): + self.configure_fixture_headless() + self.task["budget"]["max_revisions"] = 1 + self.task["budget"]["max_worker_invocations"] = 4 + started = self.service.start( + self.task, + "headless-revise", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "revise", + "reason": "Repeat the review against the same frozen candidate.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "failed") + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(receipt["accounting"]["worker_invocations"], 4) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual(len(receipt["lead"]["attempts"]), 2) + self.assertEqual(receipt["revisions"]["executed"], 1) + self.assertIn("handoff-2.json", artifacts) + self.assertIn("handoff-2.md", artifacts) + def test_invalid_internal_review_fails_before_launch_with_a_receipt(self): invalid = dict(self.fixture, verdict="clean") started = self.service.start( @@ -605,10 +806,11 @@ def test_invalid_internal_review_fails_before_launch_with_a_receipt(self): self.assertIn("verdict and findings disagree", started["error"]["message"]) result = self.service.result(started["run_id"]) self.assertTrue(result["ready"]) - self.assertEqual( - [artifact["name"] for artifact in result["artifacts"]], - ["result-receipt.json"], - ) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertEqual(set(artifacts), set(TERMINAL_REPORT_NAMES)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["phase"], "preparing") + self.assertEqual(receipt["accounting"]["worker_invocations"], 0) if __name__ == "__main__": diff --git a/test/core/test_service.py b/test/core/test_service.py index 29316fb..3e0d7c0 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -15,6 +15,7 @@ sys.path.insert(0, str(ROOT / "plugin/core/src")) from devsquad.contracts import ExecutionIdentity, LaunchSpec +from devsquad.reports import TERMINAL_REPORT_NAMES from devsquad.service import Service from devsquad.store import ConflictError, Store, canonical_json from devsquad.supervisor import Supervisor, inspect_process @@ -468,14 +469,18 @@ def test_orphan_artifact_is_not_a_result(self): try: store.finalize_artifact(started["run_id"],"orphan",b"bytes") finally: store.close() artifacts=self.service.result(started["run_id"])["artifacts"] - self.assertEqual([item["name"] for item in artifacts],["result-receipt.json"]) + self.assertEqual({item["name"] for item in artifacts},set(TERMINAL_REPORT_NAMES)) + self.assertNotIn("orphan", {item["name"] for item in artifacts}) def test_pre_attempt_failure_and_cancellation_have_durable_receipts(self): failed=self.service.start(self.task,"capability-unavailable") self.assertEqual(failed["state"],"failed") failure_result=self.service.result(failed["run_id"]) self.assertTrue(failure_result["ready"]) - failure_receipt=json.loads(Path(failure_result["artifacts"][0]["path"]).read_text()) + failure_artifact=next( + item for item in failure_result["artifacts"] if item["name"]=="receipt.json" + ) + failure_receipt=json.loads(Path(failure_artifact["path"]).read_text()) self.assertEqual(failure_receipt["state"],"failed") self.assertEqual(failure_receipt["error"]["error"],"CAPABILITY_UNAVAILABLE") @@ -673,7 +678,10 @@ def test_invalid_predecessor_fails_with_run_context_and_receipt(self): self.assertIn("superseded run",started["error"]["message"]) result=self.service.result(started["run_id"]) self.assertTrue(result["ready"]) - self.assertEqual([item["name"] for item in result["artifacts"]],["result-receipt.json"]) + self.assertEqual( + {item["name"] for item in result["artifacts"]}, + set(TERMINAL_REPORT_NAMES), + ) store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") try: self.assertIsNone(store.run(started["run_id"])["supersedes_run_id"]) @@ -725,7 +733,10 @@ def test_post_claim_snapshot_failure_has_run_context_and_receipt(self): self.assertIn("does not resolve",started["error"]["message"]) result=self.service.result(started["run_id"]) self.assertTrue(result["ready"]) - receipt=json.loads(Path(result["artifacts"][0]["path"]).read_text()) + receipt_artifact=next( + item for item in result["artifacts"] if item["name"]=="receipt.json" + ) + receipt=json.loads(Path(receipt_artifact["path"]).read_text()) self.assertEqual(receipt["run_id"],started["run_id"]) self.assertEqual(receipt["error"],started["error"]) diff --git a/test/core/test_store.py b/test/core/test_store.py index c2e06ca..04d0198 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -231,8 +231,8 @@ def test_artifact_is_finalized_and_verified_before_reference(self): self.assertEqual(row[1], 13) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 5) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(6,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 6) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(7,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -247,8 +247,12 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 5) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 6) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) + attempt_columns = { + row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") + } + self.assertIn("role", attempt_columns) def test_version_three_fixture_adds_run_snapshot_columns(self): old_db=self.root/"v3.sqlite3"; connection=sqlite3.connect(old_db) @@ -257,7 +261,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],5) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],6) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From e1afa4dc3ad761f966e575fefd009b7c49147eed Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 18 Sep 2026 14:08:02 +0530 Subject: [PATCH 063/197] WIP checkpoint: fix: enforce M3 runtime budgets and pool capacity (2026-09-18 14:08) --- plugin/core/src/devsquad/contracts.py | 4 + plugin/core/src/devsquad/detached.py | 15 +- .../migrations/007_attempt_account_pools.sql | 5 + plugin/core/src/devsquad/reports.py | 10 +- plugin/core/src/devsquad/router.py | 23 ++ plugin/core/src/devsquad/service.py | 254 ++++++++++++++++-- plugin/core/src/devsquad/store.py | 236 +++++++++++++++- plugin/core/src/devsquad/supervisor.py | 15 +- test/core/test_cli.py | 6 +- test/core/test_handoff_store.py | 6 +- test/core/test_review_runtime.py | 77 ++++++ test/core/test_store.py | 134 ++++++++- 12 files changed, 729 insertions(+), 56 deletions(-) create mode 100644 plugin/core/src/devsquad/migrations/007_attempt_account_pools.sql diff --git a/plugin/core/src/devsquad/contracts.py b/plugin/core/src/devsquad/contracts.py index fbb132e..0f8a820 100644 --- a/plugin/core/src/devsquad/contracts.py +++ b/plugin/core/src/devsquad/contracts.py @@ -25,6 +25,10 @@ class CapabilityUnavailable(ContractError): code = "CAPABILITY_UNAVAILABLE" +class BudgetExhausted(ContractError): + code = "BUDGET_EXHAUSTED" + + class PolicyDenied(ContractError): code = "POLICY_DENIED" diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index bd44104..0591b09 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -5,7 +5,7 @@ from pathlib import Path import sys -from .contracts import ExecutionIdentity, LaunchSpec +from .contracts import BudgetExhausted, ExecutionIdentity, LaunchSpec from .service import Service from .store import ConflictError, Store, canonical_json from .supervisor import Supervisor @@ -80,6 +80,12 @@ def main(argv=None): command = [sys.executable, "-P", "-m", "devsquad.fake_step"] if "internal_fake_delay" in snapshot: command += ["--delay", str(snapshot["internal_fake_delay"])] + remaining_wall = store.remaining_wall_seconds(args.run_id) + if remaining_wall == 0: + Service(Path(args.database).parent).fail_budget_exhausted( + args.run_id, args.expected_version, + ) + return 1 spec = LaunchSpec( 1, identity.harness, @@ -87,13 +93,18 @@ def main(argv=None): tuple(command), run["worktree_path"], stdin_path, - snapshot["task"]["budget"]["wall_seconds"], + remaining_wall or snapshot["task"]["budget"]["wall_seconds"], identity, environment, ) supervisor = Supervisor(store) try: handle = supervisor.launch_durable(args.run_id, args.expected_version, spec, f"daemon:{os.getpid()}", args.package_digest, role=role if workflow_role else "worker") except ConflictError: return 0 + except BudgetExhausted: + Service(Path(args.database).parent).fail_budget_exhausted( + args.run_id, args.expected_version, + ) + return 1 returncode = supervisor.wait_durable(handle, spec.timeout_seconds) current = store.run(args.run_id) finally: diff --git a/plugin/core/src/devsquad/migrations/007_attempt_account_pools.sql b/plugin/core/src/devsquad/migrations/007_attempt_account_pools.sql new file mode 100644 index 0000000..b740081 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/007_attempt_account_pools.sql @@ -0,0 +1,5 @@ +ALTER TABLE attempts ADD COLUMN account_pool_id TEXT; + +CREATE INDEX attempts_active_account_pool +ON attempts(account_pool_id, status) +WHERE account_pool_id IS NOT NULL; diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index d3ffc67..8a60a8e 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -286,7 +286,11 @@ def build_early_terminal_reports( "dispositions": [], "lead": { "mode": task.get("lead", {}).get("mode"), - "status": "failed" if phase == "lead" else "not_reached", + "status": ( + "cancelled" + if state == "cancelled" and phase in {"lead", "awaiting_host"} + else "failed" if phase == "lead" else "not_reached" + ), "disposition": None, "reason": None, "usage": { @@ -316,7 +320,9 @@ def build_early_terminal_reports( "excludes_terminal_report_artifact_events": True, }, "limitations": [( - "The headless lead failed before a valid disposition was recorded." + "The run was cancelled before a lead disposition became terminal." + if state == "cancelled" + else "The headless lead failed before a valid disposition was recorded." if phase == "lead" else "No valid review handoff was produced, so lead disposition was not reached." )], diff --git a/plugin/core/src/devsquad/router.py b/plugin/core/src/devsquad/router.py index 5ed13f8..7155046 100644 --- a/plugin/core/src/devsquad/router.py +++ b/plugin/core/src/devsquad/router.py @@ -388,3 +388,26 @@ def load_routing( profiles_sha256=profiles_sha256, policy_sha256=policy_sha256, ) + + +def capacity_with_live_reservations( + policy_payload: bytes | str, + in_flight: dict[str, int], +) -> dict[str, dict[str, Any]]: + """Bind transactionally observed local reservations to one policy snapshot.""" + policy, _ = _strict_json(policy_payload, "policy file") + validate_policy(policy) + if (not isinstance(in_flight, dict) + or not all( + isinstance(pool_id, str) and pool_id + and type(count) is int and count >= 0 + for pool_id, count in in_flight.items() + )): + raise ContractError("live capacity reservations are invalid") + return { + pool_id: { + "status": "unknown", + "in_flight": in_flight.get(pool_id, 0), + } + for pool_id in policy["account_pools"] + } diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 45e0067..e1666d8 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -15,9 +15,14 @@ from .codex_lead_worker import freeze_codex_lead from .codex_review_worker import freeze_codex_reviewer -from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported +from .contracts import ( + BudgetExhausted, + CapabilityUnavailable, + ContractError, + ProfileUnsupported, +) from .reports import build_early_terminal_reports, build_terminal_reports -from .router import load_routing +from .router import capacity_with_live_reservations, load_routing from .store import ( ConflictError, HandoffClaim, @@ -88,6 +93,164 @@ def _preparation_failure_artifacts( }) return prepared + def _paused_review_terminal_artifacts( + self, + store: Store, + run_id: str, + snapshot: dict[str, Any], + handoff: HandoffSnapshot | None, + *, + state: str, + phase: str, + error: dict[str, Any] | None, + ) -> list[dict[str, Any]]: + """Materialize complete M3 reports before a paused run terminalizes.""" + history = store.branch_review_history(run_id) + frozen_handoffs = { + item["handoff_id"]: { + "handoff_id": item["handoff_id"], + "packet": item["packet"], + "packet_sha256": item["packet_sha256"], + } + for item in history + } + if handoff is not None: + frozen_handoffs[handoff.handoff_id] = { + "handoff_id": handoff.handoff_id, + "packet": handoff.packet, + "packet_sha256": handoff.packet_sha256, + } + packets = [item["packet"] for item in history] + if (handoff is not None and not any( + packet.get("attempt_id") == handoff.packet.get("attempt_id") + for packet in packets)): + packets.append(handoff.packet) + packets_by_attempt = { + packet["attempt_id"]: validate_branch_review_handoff(packet, snapshot) + for packet in packets + } + artifacts = store.artifacts_for_run(run_id) + artifacts_by_name = {artifact["name"]: artifact for artifact in artifacts} + attempts = [] + for attempt in store.attempts_for_run(run_id): + if attempt.get("role") == "reviewer": + packet = packets_by_attempt.get(attempt["id"]) + if packet is not None: + attempts.append({ + "id": attempt["id"], + "status": "succeeded", + **packet["attempt"], + "review": packet["review"], + "checks": packet["checks"], + "evaluation": packet["evaluation"], + }) + elif attempt.get("role") == "lead" and attempt.get("status") == "finished": + artifact = artifacts_by_name.get( + f"lead-attempt-{attempt['id']}.json" + ) + if artifact is None: + raise ConflictError("saved headless lead evidence is missing") + content = Path(artifact["path"]).read_bytes() + try: + raw = json.loads(content) + frozen_handoff = frozen_handoffs[raw["handoff_id"]] + except (KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError( + "saved headless lead evidence is invalid" + ) from exc + evidence = decode_headless_lead_evidence( + content, snapshot, frozen_handoff, + ) + attempts.append({ + "id": attempt["id"], + "status": "succeeded", + **evidence["attempt"], + }) + reports = build_early_terminal_reports( + run_id=run_id, + state=state, + task=snapshot["task"], + snapshot=snapshot, + run_artifacts=artifacts, + events=store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + phase=phase, + error=error, + prior_attempts=attempts, + ) + prepared = [] + for name in sorted(reports): + path, digest, size = store.finalize_artifact( + run_id, name, reports[name], + ) + prepared.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + return prepared + + def fail_budget_exhausted( + self, + run_id: str, + expected_version: int, + ) -> dict[str, Any]: + """Terminalize a queued branch review that cannot launch another worker.""" + store = self._store() + try: + run = store.run(run_id) + snapshot = self._review_snapshot(run) + if snapshot.get("task", {}).get("workflow") != "branch-review": + raise ConflictError("queued budget failure is not a branch review") + error = { + "error": "BUDGET_EXHAUSTED", + "message": "run wall-time budget is exhausted", + } + history = store.branch_review_history(run_id) + if history: + run_artifacts = store.artifacts_for_run(run_id) + reports = build_terminal_reports( + run_id=run_id, + state="failed", + snapshot=snapshot, + history=history, + run_artifacts=run_artifacts, + events=store.events_for_run(run_id), + completed_at=datetime.now(timezone.utc).isoformat(), + error=error, + headless_leads=self._headless_leads_for_history( + snapshot, history, run_artifacts, + ), + ) + terminal_artifacts = [] + for name in sorted(reports): + path, digest, size = store.finalize_artifact( + run_id, name, reports[name], + ) + terminal_artifacts.append({ + "name": name, + "path": path, + "sha256": digest, + "byte_size": size, + }) + else: + terminal_artifacts = self._paused_review_terminal_artifacts( + store, + run_id, + snapshot, + store.handoff_snapshot(run_id), + state="failed", + phase="budget", + error=error, + ) + version = store.fail_queued_budget( + run_id, expected_version, error, terminal_artifacts, + ) + return {"run_id": run_id, "state": "failed", "version": version} + finally: + store.close() + def _freeze_package(self) -> tuple[Path, str]: source = Path(__file__).resolve().parent digest = hashlib.sha256() @@ -209,6 +372,7 @@ def _resolve_snapshot( run_id: str | None = None, internal_review_fixture: dict[str, Any] | None = None, internal_lead_fixture: dict[str, Any] | None = None, + capacity_in_flight: dict[str, int] | None = None, ) -> dict[str, Any]: repo = resolved_repo or Path(task["project"]["repo_path"]).resolve(strict=True) base_oid = resolve_commit(repo, task["project"]["base_ref"]) @@ -248,6 +412,9 @@ def _resolve_snapshot( task, config_payloads["profiles_file"], config_payloads["policy_file"], + availability=capacity_with_live_reservations( + config_payloads["policy_file"], capacity_in_flight or {}, + ), ) if project_id is None or run_id is None: raise ContractError("public preflight requires run-owned workspace identity") @@ -334,6 +501,7 @@ def _continue_preparation( run_id=run_id, internal_review_fixture=internal_review_fixture, internal_lead_fixture=internal_lead_fixture, + capacity_in_flight=store.active_pool_counts(), ) if internal_delay is None and internal_review_fixture is None: snapshot["review_adapter"] = freeze_codex_reviewer( @@ -344,6 +512,8 @@ def _continue_preparation( snapshot["lead_adapter"] = freeze_codex_lead( snapshot["routing"]["roles"]["lead"]["selected"], ) + if store.remaining_wall_seconds(run_id) == 0: + raise BudgetExhausted("run wall-time budget is exhausted in preflight") package, digest = self._freeze_package() version = store.complete_preparation( run_id, @@ -355,7 +525,7 @@ def _continue_preparation( worktree_path=(snapshot.get("workspace") or {}).get("path"), ) return (version, package, digest), None - except (CapabilityUnavailable, ProfileUnsupported) as exc: + except (BudgetExhausted, CapabilityUnavailable, ProfileUnsupported) as exc: error = {"error": exc.code, "message": str(exc)} terminal_artifacts = self._preparation_failure_artifacts( store, run_id, task, snapshot, error, @@ -504,7 +674,27 @@ def cancel(self, run_id: str) -> dict[str, Any]: from .supervisor import Supervisor version = Supervisor(store).cancel_orphan(run_id) elif run["state"] in {"running", "cancelling"}: version, _ = store.request_cancel(run_id) - elif run["state"] == "awaiting_host": version = store.cancel_host_wait(run_id) + elif run["state"] == "awaiting_host": + terminal_artifacts = None + try: + snapshot = self._review_snapshot(run) + handoff = store.handoff_snapshot(run_id) + if (snapshot.get("task", {}).get("workflow") == "branch-review" + and handoff is not None): + terminal_artifacts = self._paused_review_terminal_artifacts( + store, + run_id, + snapshot, + handoff, + state="cancelled", + phase="awaiting_host", + error=None, + ) + except (KeyError, TypeError): + terminal_artifacts = None + version = store.cancel_host_wait( + run_id, terminal_artifacts=terminal_artifacts, + ) elif run["state"] == "blocked": from .supervisor import Supervisor version = Supervisor(store).cancel_orphan(run_id) @@ -586,6 +776,36 @@ def _review_gate( ) return packet, gate + @staticmethod + def _headless_leads_for_history( + snapshot: dict[str, Any], + history: list[dict[str, Any]], + run_artifacts: list[dict[str, Any]], + ) -> list[dict[str, Any]]: + if snapshot["task"]["lead"]["mode"] != "headless": + return [] + artifacts_by_name = { + artifact["name"]: artifact for artifact in run_artifacts + } + evidence = [] + for item in history: + submission_id = item["decision"]["submission_id"] + if not submission_id.startswith("headless-"): + raise ConflictError("headless lead submission identity is invalid") + attempt_id = submission_id[len("headless-"):] + artifact = artifacts_by_name.get(f"lead-attempt-{attempt_id}.json") + if artifact is None: + raise ConflictError("headless lead evidence artifact is missing") + frozen_handoff = { + "handoff_id": item["handoff_id"], + "packet": item["packet"], + "packet_sha256": item["packet_sha256"], + } + evidence.append(decode_headless_lead_evidence( + Path(artifact["path"]).read_bytes(), snapshot, frozen_handoff, + )) + return evidence + def _terminalize_branch_review( self, store: Store, @@ -614,29 +834,9 @@ def _terminalize_branch_review( or hashlib.sha256(path.read_bytes()).hexdigest() != artifact["sha256"]): raise ConflictError("branch review artifact is missing or corrupt") - headless_leads = [] - if snapshot["task"]["lead"]["mode"] == "headless": - artifacts_by_name = { - artifact["name"]: artifact for artifact in run_artifacts - } - for item in history: - submission_id = item["decision"]["submission_id"] - if not submission_id.startswith("headless-"): - raise ConflictError("headless lead submission identity is invalid") - attempt_id = submission_id[len("headless-"):] - artifact = artifacts_by_name.get( - f"lead-attempt-{attempt_id}.json" - ) - if artifact is None: - raise ConflictError("headless lead evidence artifact is missing") - frozen_handoff = { - "handoff_id": item["handoff_id"], - "packet": item["packet"], - "packet_sha256": item["packet_sha256"], - } - headless_leads.append(decode_headless_lead_evidence( - Path(artifact["path"]).read_bytes(), snapshot, frozen_handoff, - )) + headless_leads = self._headless_leads_for_history( + snapshot, history, run_artifacts, + ) reports = build_terminal_reports( run_id=run_id, state=terminal_state, diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 28bbcb2..8715b7e 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -15,9 +15,9 @@ import uuid from typing import Any -from .contracts import ContractError +from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 6 +SUPPORTED_SCHEMA_VERSION = 7 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -596,6 +596,104 @@ def store_artifact(self, run_id: str, name: str, content: bytes) -> str: path, digest, _ = self.finalize_artifact(run_id, name, content) return self.reference_artifact(run_id, name, path, digest) + @staticmethod + def _wall_seconds_from_run(run: sqlite3.Row | dict[str, Any]) -> int | None: + documents = (run.get("mutable_snapshot"), run.get("submitted_request")) \ + if isinstance(run, dict) else (run["mutable_snapshot"], run["submitted_request"]) + for encoded in documents: + if not encoded: + continue + try: + document = json.loads(encoded) + task = document.get("task") if isinstance(document, dict) else None + budget = task.get("budget") if isinstance(task, dict) else None + wall_seconds = budget.get("wall_seconds") if isinstance(budget, dict) else None + except (TypeError, json.JSONDecodeError): + continue + if type(wall_seconds) is int and wall_seconds > 0: + return wall_seconds + return None + + def _execution_elapsed_ms( + self, + run_id: str, + run: sqlite3.Row | dict[str, Any], + current: datetime, + ) -> int: + """Count preflight and worker intervals while excluding saved host waits.""" + now = current.astimezone(timezone.utc) + created_at = _parse_utc(run["created_at"]) + queued = self.connection.execute( + "SELECT created_at FROM events WHERE run_id=? AND type='run.queued' " + "ORDER BY id LIMIT 1", + (run_id,), + ).fetchone() + if queued is not None: + preflight_end = _parse_utc(queued["created_at"]) + elif run["state"] == "queued" and run["phase"] == "preparing": + preflight_end = now + else: + preflight_end = _parse_utc(run["updated_at"]) + elapsed = max(0, int((preflight_end - created_at).total_seconds() * 1000)) + for attempt in self.connection.execute( + "SELECT status,created_at,finished_at FROM attempts WHERE run_id=?", + (run_id,), + ): + started = _parse_utc(attempt["created_at"]) + if (attempt["status"] == "ownership_ambiguous" + or attempt["finished_at"] is None): + finished = now + else: + finished = _parse_utc(attempt["finished_at"]) + elapsed += max(0, int((finished - started).total_seconds() * 1000)) + return elapsed + + def remaining_wall_seconds( + self, + run_id: str, + *, + now: datetime | None = None, + ) -> int | None: + run = self.connection.execute( + "SELECT state,phase,created_at,updated_at,mutable_snapshot,submitted_request " + "FROM runs WHERE id=?", + (run_id,), + ).fetchone() + if run is None: + raise ContractError("run does not exist") + wall_seconds = self._wall_seconds_from_run(run) + if wall_seconds is None: + return None + elapsed_ms = self._execution_elapsed_ms( + run_id, run, _authoritative_now(now), + ) + return max(0, (wall_seconds * 1000 - elapsed_ms) // 1000) + + def _enforce_attempt_budget( + self, + run_id: str, + run: sqlite3.Row, + ) -> None: + wall_seconds = self._wall_seconds_from_run(run) + if wall_seconds is None: + return + elapsed_ms = self._execution_elapsed_ms( + run_id, run, _authoritative_now(), + ) + if wall_seconds * 1000 - elapsed_ms < 1000: + raise BudgetExhausted("run wall-time budget is exhausted") + + def active_pool_counts(self) -> dict[str, int]: + return { + row["account_pool_id"]: row["in_flight"] + for row in self.connection.execute( + "SELECT account_pool_id,COUNT(*) AS in_flight FROM attempts " + "WHERE account_pool_id IS NOT NULL AND status IN " + "('reserved','running','cancelling','ownership_ambiguous') " + "GROUP BY account_pool_id" + ) + } + def reserve_attempt( self, run_id: str, @@ -603,6 +701,8 @@ def reserve_attempt( owner_id: str, package_digest: str, role: str = "worker", + *, + account_pool_id: str | None = None, ) -> AttemptReservation: if not owner_id or not package_digest: raise ContractError("supervisor owner and package digest are required") @@ -610,7 +710,11 @@ def reserve_attempt( raise ContractError("attempt role is invalid") self.connection.execute("BEGIN IMMEDIATE") try: - run = self.connection.execute("SELECT project_id,worktree_path,state,phase,version FROM runs WHERE id=?", (run_id,)).fetchone() + run = self.connection.execute( + "SELECT project_id,worktree_path,state,phase,version,created_at,updated_at," + "mutable_snapshot,submitted_request FROM runs WHERE id=?", + (run_id,), + ).fetchone() if not run or run["version"] != expected_version or run["state"] != "queued" or run["phase"] is not None: raise ConflictError("run is not available for supervisor claim") active = self.connection.execute("SELECT 1 FROM supervisor_claims WHERE run_id=? AND active=1", (run_id,)).fetchone() @@ -618,11 +722,44 @@ def reserve_attempt( raise ConflictError("run already has a supervisor claim") if not run["worktree_path"]: raise ContractError("run has no canonical worktree identity") + self._enforce_attempt_budget(run_id, run) + if account_pool_id is not None: + if not isinstance(account_pool_id, str) or not account_pool_id: + raise ContractError("attempt account pool is invalid") + try: + snapshot = json.loads(run["mutable_snapshot"]) + selected = snapshot["routing"]["roles"][role]["selected"]["profile"] + capacity = snapshot["routing"]["capacity"][account_pool_id] + expected_pool = selected["account_pool_id"] + max_concurrency = capacity["max_concurrency"] + except (KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError( + "frozen account-pool reservation is invalid" + ) from exc + if (expected_pool != account_pool_id + or type(max_concurrency) is not int + or max_concurrency < 1): + raise ConflictError( + "attempt account pool does not match frozen routing" + ) + in_flight = self.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE account_pool_id=? " + "AND status IN ('reserved','running','cancelling','ownership_ambiguous')", + (account_pool_id,), + ).fetchone()[0] + if in_flight >= max_concurrency: + raise ConflictError("account pool concurrency is full") old = self.connection.execute("SELECT COALESCE(MAX(fencing_token),0) FROM supervisor_claims WHERE run_id=?", (run_id,)).fetchone()[0] supervisor_token, attempt_id, attempt_token = old + 1, str(uuid.uuid4()), uuid.uuid4().hex now, version = _utc_now(), expected_version + 1 self.connection.execute("INSERT OR REPLACE INTO supervisor_claims(run_id,owner_id,fencing_token,package_digest,heartbeat_at,active) VALUES(?,?,?,?,?,1)", (run_id, owner_id, supervisor_token, package_digest, now)) - self.connection.execute("INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token,status,heartbeat_at,package_digest,created_at,role) VALUES(?,?,?,?,?,'reserved',?,?,?,?)", (attempt_id, run_id, run["project_id"], run["worktree_path"], attempt_token, now, package_digest, now, role)) + self.connection.execute( + "INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token," + "status,heartbeat_at,package_digest,created_at,role,account_pool_id) " + "VALUES(?,?,?,?,?,'reserved',?,?,?,?,?)", + (attempt_id, run_id, run["project_id"], run["worktree_path"], + attempt_token, now, package_digest, now, role, account_pool_id), + ) self.connection.execute("UPDATE runs SET phase='launching',version=?,updated_at=? WHERE id=?", (version, now, run_id)) event = canonical_json({"attempt_id": attempt_id, "supervisor_token": supervisor_token}) self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'supervisor.claimed',?,?)", (run_id, version, event, now)) @@ -2269,7 +2406,19 @@ def complete_handoff_terminal( self.connection.execute("ROLLBACK") raise - def cancel_host_wait(self, run_id: str, *, now: datetime | None = None) -> int: + def cancel_host_wait( + self, + run_id: str, + *, + now: datetime | None = None, + terminal_artifacts: list[dict[str, Any]] | None = None, + ) -> int: + prepared = ( + self._prepare_exact_artifacts( + run_id, terminal_artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS, + ) + if terminal_artifacts is not None else None + ) self.connection.execute("BEGIN IMMEDIATE") try: timestamp = _authoritative_now(now).isoformat() @@ -2289,12 +2438,18 @@ def cancel_host_wait(self, run_id: str, *, now: datetime | None = None) -> int: or handoff["status"] not in {"open", "submitted"} or run["phase"] not in {None, "handoff_submitted"}): raise ConflictError("run is not awaiting a cancellable host handoff") - path, digest, size, receipt_time = self._terminal_receipt( - run_id, "cancelled", "awaiting_host", None, now=timestamp, - ) - version = self._reference_terminal_receipt( - run_id, run["version"], path, digest, size, receipt_time, - ) + 1 + if prepared is None: + path, digest, size, receipt_time = self._terminal_receipt( + run_id, "cancelled", "awaiting_host", None, now=timestamp, + ) + version = self._reference_terminal_receipt( + run_id, run["version"], path, digest, size, receipt_time, + ) + else: + version, _ = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version += 1 self.connection.execute( "UPDATE handoffs SET status='cancelled',closed_at=? WHERE id=?", (timestamp, handoff["id"]), @@ -2309,7 +2464,9 @@ def cancel_host_wait(self, run_id: str, *, now: datetime | None = None) -> int: (version, timestamp, run_id), ) payload = canonical_json({ - "handoff_id": handoff["id"], "receipt": "result-receipt.json", + "handoff_id": handoff["id"], + "receipt": "result-receipt.json", + "terminal_reports": prepared is not None, }) self.connection.execute( "INSERT INTO events(run_id,run_version,type,payload,created_at) " @@ -2322,6 +2479,61 @@ def cancel_host_wait(self, run_id: str, *, now: datetime | None = None) -> int: self.connection.execute("ROLLBACK") raise + def fail_queued_budget( + self, + run_id: str, + expected_version: int, + error: dict[str, Any], + terminal_artifacts: list[dict[str, Any]], + ) -> int: + """Fail a paused workflow before another worker launch can consume budget.""" + prepared = self._prepare_exact_artifacts( + run_id, terminal_artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS, + ) + encoded_error = canonical_json(error) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + if (not run or run["state"] != "queued" or run["phase"] is not None + or run["version"] != expected_version): + raise ConflictError("budget exhaustion is no longer current") + if self.connection.execute( + "SELECT 1 FROM supervisor_claims WHERE run_id=? AND active=1", + (run_id,), + ).fetchone(): + raise ConflictError("budget exhaustion raced with a supervisor") + version, _ = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE handoffs SET status='consumed',closed_at=? WHERE run_id=? " + "AND status IN ('open','submitted')", + (now, run_id), + ) + self.connection.execute( + "UPDATE claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='failed',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = dict(error) + payload["receipt"] = "result-receipt.json" + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.failed',?,?)", + (run_id, version, canonical_json(payload), now), + ) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + def cancel_queued(self, run_id: str) -> int: self.connection.execute("BEGIN IMMEDIATE") try: diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 1107c0f..74598f0 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -166,7 +166,13 @@ def __init__(self, store: Store, *, output_limit: int = 1024 * 1024, grace_secon self.store, self.output_limit, self.grace_seconds = store, output_limit, grace_seconds def launch(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: str, package_digest: str) -> RunningAttempt: - reservation = self.store.reserve_attempt(run_id, expected_version, owner_id, package_digest) + reservation = self.store.reserve_attempt( + run_id, + expected_version, + owner_id, + package_digest, + account_pool_id=spec.requested.account_pool, + ) environment = os.environ.copy(); environment.update(spec.environment) try: stdin_stream: Any = subprocess.DEVNULL @@ -220,7 +226,12 @@ def launch_durable( role: str = "worker", ) -> DurableAttempt: reservation = self.store.reserve_attempt( - run_id, expected_version, owner_id, package_digest, role, + run_id, + expected_version, + owner_id, + package_digest, + role, + account_pool_id=spec.requested.account_pool, ) directory = self.store.artifacts / run_id / f".{reservation.attempt_id}.spool" directory.mkdir(parents=True, exist_ok=False) diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 20d1681..a53d7bf 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -291,7 +291,7 @@ def build_python(): return candidate return None - def test_installed_wheel_contains_and_applies_migrations_through_six(self): + def test_installed_wheel_contains_and_applies_migrations_through_seven(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") @@ -333,9 +333,9 @@ def test_installed_wheel_contains_and_applies_migrations_through_six(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 6 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 7 attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} - assert "role" in attempt_columns + assert {"role", "account_pool_id"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} assert {"package_path", "package_digest", "supersedes_run_id"} <= columns assert store.connection.execute( diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 2ac45ac..3a86559 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 6, + 7, ) tables = { row[0] @@ -618,7 +618,7 @@ def build_python(): return candidate return None - def test_installed_wheel_applies_schema_four_to_six(self): + def test_installed_wheel_applies_schema_four_to_seven(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel") @@ -684,7 +684,7 @@ def test_installed_wheel_applies_schema_four_to_six(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 6 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 7 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index 07ae352..e3903b5 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -388,6 +388,49 @@ def test_zero_revision_budget_turns_revise_into_terminal_failure(self): self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") self.assertEqual(receipt["lead"]["disposition"], "revise") + def test_wall_budget_exhaustion_between_attempts_preserves_review_history(self): + self.task["budget"]["max_revisions"] = 1 + self.task["budget"]["max_worker_invocations"] = 2 + run_id, waiting = self.start_waiting("wall-budget-between-attempts") + claimed = self.service.handoff_claim( + run_id, waiting["version"], "host-wall-budget", + ) + packet = claimed["handoff"]["packet"] + revise = self.decision( + packet, "wall-budget-revise", "revise", "Repeat once.", + ) + with patch.object(self.service, "_spawn_daemon", return_value=12345): + requeued = self.service.handoff_complete( + run_id, claimed["claim"], revise, + ) + self.assertEqual(requeued["state"], "queued") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + attempt = store.attempt(run_id) + old = "2026-09-18T00:00:00+00:00" + finished = "2026-09-18T00:10:00+00:00" + store.connection.execute( + "UPDATE attempts SET created_at=?,finished_at=? WHERE id=?", + (old, finished, attempt["id"]), + ) + finally: + store.close() + terminal = self.service.fail_budget_exhausted( + run_id, requeued["version"], + ) + self.assertEqual(terminal["state"], "failed") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(run_id)["artifacts"] + } + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(len(receipt["attempts"]), 1) + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], ["revise"], + ) + def test_resume_finishes_submission_recorded_before_continuation(self): run_id, waiting = self.start_waiting("resume-submission") claimed = self.service.handoff_claim(run_id, waiting["version"], "host-crash") @@ -711,6 +754,40 @@ def test_waiting_handoff_report_builder_binds_packet_and_hash(self): self.assertEqual(document["packet_sha256"], handoff.packet_sha256) self.assertIn(b"claim this saved handoff", reports["handoff.md"]) + def test_cancel_waiting_review_publishes_full_terminal_reports(self): + run_id, _ = self.start_waiting("cancel-waiting-review") + cancelled = self.service.cancel(run_id) + self.assertEqual(cancelled["state"], "cancelled") + result = self.service.result(run_id) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["state"], "cancelled") + self.assertEqual(receipt["phase"], "awaiting_host") + self.assertEqual(receipt["lead"]["status"], "cancelled") + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + + def test_public_preflight_observes_live_shared_pool_reservations(self): + with patch.object( + Store, "active_pool_counts", + return_value={"fixture-subscription": 1}, + ): + started = self.service.start( + self.task, + "pool-full-before-launch", + _internal_review_fixture=self.fixture, + ) + self.assertEqual(started["state"], "failed") + self.assertEqual(started["error"]["error"], "CAPABILITY_UNAVAILABLE") + self.assertIn("no currently available profile", started["error"]["message"]) + result = self.service.result(started["run_id"]) + self.assertTrue(result["ready"]) + self.assertTrue( + TERMINAL_REPORT_NAMES + <= {artifact["name"] for artifact in result["artifacts"]} + ) + def test_headless_lead_is_a_second_fenced_attempt_and_terminalizes_automatically(self): self.configure_fixture_headless() started = self.service.start( diff --git a/test/core/test_store.py b/test/core/test_store.py index 04d0198..6d69664 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -1,5 +1,6 @@ import hashlib import json +from datetime import datetime, timedelta, timezone from pathlib import Path import sqlite3 import subprocess @@ -11,7 +12,7 @@ import sys sys.path.insert(0, str(ROOT / "plugin/core/src")) -from devsquad.contracts import ContractError +from devsquad.contracts import BudgetExhausted, ContractError from devsquad.store import ConflictError, SchemaVersionError, Store, git_common_dir @@ -230,9 +231,131 @@ def test_artifact_is_finalized_and_verified_before_reference(self): row = self.store.connection.execute("SELECT sha256,byte_size FROM artifacts WHERE id=?", (artifact_id,)).fetchone() self.assertEqual(row[1], 13) + @staticmethod + def routed_snapshot(wall_seconds=300): + return { + "task": {"budget": {"wall_seconds": wall_seconds}}, + "routing": { + "roles": { + "reviewer": { + "selected": { + "profile": {"account_pool_id": "shared-pool"}, + }, + }, + }, + "capacity": { + "shared-pool": {"max_concurrency": 1}, + }, + }, + } + + def test_shared_pool_reservation_is_transactional_and_releases_on_finish(self): + linked = self.root / "pool-linked" + subprocess.run( + ["git", "-C", str(self.repo), "worktree", "add", "--detach", "-q", + str(linked), "HEAD"], + check=True, + ) + first = self.store.claim_start( + self.repo, "pool-first", {"task": {"budget": {"wall_seconds": 300}}}, + "owner", + ) + first_version = self.store.complete_preparation( + first.run_id, + first.fencing_token, + self.routed_snapshot(), + worktree_path=str(self.repo), + ) + second = self.store.claim_start( + linked, "pool-second", {"task": {"budget": {"wall_seconds": 300}}}, + "owner", + ) + second_version = self.store.complete_preparation( + second.run_id, + second.fencing_token, + self.routed_snapshot(), + worktree_path=str(linked), + ) + reservation = self.store.reserve_attempt( + first.run_id, + first_version, + "supervisor-one", + "package", + "reviewer", + account_pool_id="shared-pool", + ) + self.assertEqual(self.store.active_pool_counts(), {"shared-pool": 1}) + with self.assertRaisesRegex(ConflictError, "account pool concurrency"): + self.store.reserve_attempt( + second.run_id, + second_version, + "supervisor-two", + "package", + "reviewer", + account_pool_id="shared-pool", + ) + self.store.mark_attempt_running(reservation, 1001, 1001, "fixture-process") + self.store.finish_attempt( + first.run_id, reservation.attempt_token, "succeeded", {}, + ) + self.assertEqual(self.store.active_pool_counts(), {}) + released = self.store.reserve_attempt( + second.run_id, + second_version, + "supervisor-two", + "package", + "reviewer", + account_pool_id="shared-pool", + ) + self.assertEqual(released.run_id, second.run_id) + + def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): + claim = self.store.claim_start( + self.repo, + "wall-budget", + {"task": {"budget": {"wall_seconds": 5}}}, + "owner", + ) + version = self.store.complete_preparation( + claim.run_id, claim.fencing_token, self.routed_snapshot(5), + ) + started = datetime(2026, 9, 18, 1, 0, tzinfo=timezone.utc) + self.store.connection.execute( + "UPDATE runs SET created_at=?,updated_at=? WHERE id=?", + (started.isoformat(), (started + timedelta(seconds=5)).isoformat(), + claim.run_id), + ) + self.store.connection.execute( + "UPDATE events SET created_at=? WHERE run_id=? AND type='run.preparing'", + (started.isoformat(), claim.run_id), + ) + self.store.connection.execute( + "UPDATE events SET created_at=? WHERE run_id=? AND type='run.queued'", + ((started + timedelta(milliseconds=200)).isoformat(), claim.run_id), + ) + run = self.store.run(claim.run_id) + self.store.connection.execute( + "INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token," + "status,heartbeat_at,package_digest,created_at,finished_at,role,account_pool_id) " + "VALUES('prior-attempt',?,?,?,?, 'finished',?,?,?,?, 'reviewer','shared-pool')", + (claim.run_id, run["project_id"], run["worktree_path"], "prior-token", + (started + timedelta(seconds=5)).isoformat(), "package", + (started + timedelta(seconds=1)).isoformat(), + (started + timedelta(seconds=5, milliseconds=100)).isoformat()), + ) + current = started + timedelta(seconds=6) + self.assertEqual( + self.store.remaining_wall_seconds(claim.run_id, now=current), 0, + ) + with self.assertRaises(BudgetExhausted): + self.store.reserve_attempt( + claim.run_id, version, "supervisor", "package", "reviewer", + account_pool_id="shared-pool", + ) + def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 6) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(7,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 7) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(8,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -247,12 +370,13 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 6) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 7) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") } self.assertIn("role", attempt_columns) + self.assertIn("account_pool_id", attempt_columns) def test_version_three_fixture_adds_run_snapshot_columns(self): old_db=self.root/"v3.sqlite3"; connection=sqlite3.connect(old_db) @@ -261,7 +385,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],6) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],7) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From 84deb7772af46581331cf716b0f5dcbd693fe0f6 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sat, 19 Sep 2026 00:40:16 +0530 Subject: [PATCH 064/197] feat: execute bounded frozen fallbacks --- plugin/core/src/devsquad/detached.py | 86 +++++++- plugin/core/src/devsquad/lead_worker.py | 3 + .../migrations/008_attempt_routing.sql | 2 + plugin/core/src/devsquad/reports.py | 24 ++- plugin/core/src/devsquad/review_worker.py | 3 + plugin/core/src/devsquad/service.py | 86 +++++++- plugin/core/src/devsquad/store.py | 130 +++++++++++-- plugin/core/src/devsquad/supervisor.py | 99 +++++++++- plugin/core/src/devsquad/workflows.py | 56 ++++-- test/core/test_cli.py | 6 +- test/core/test_handoff_store.py | 8 +- test/core/test_review_runtime.py | 183 ++++++++++++++++++ test/core/test_review_workflow.py | 2 +- test/core/test_store.py | 10 +- 14 files changed, 639 insertions(+), 59 deletions(-) create mode 100644 plugin/core/src/devsquad/migrations/008_attempt_routing.sql diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index 0591b09..9beaa89 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -11,6 +11,37 @@ from .supervisor import Supervisor +def _profile_index( + store: Store, + run_id: str, + role: str, + handoff, +) -> int: + attempts = store.attempts_for_run(run_id) + if handoff is None: + return sum(attempt.get("role") == role for attempt in attempts) + reviewer_id = handoff.packet.get("attempt_id") + reviewer = next( + (attempt for attempt in attempts if attempt["id"] == reviewer_id), None, + ) + if reviewer is None: + raise ConflictError("handoff reviewer attempt is missing") + if role == "reviewer": + seen = False + used = 0 + for attempt in attempts: + if attempt["id"] == reviewer_id: + seen = True + elif seen and attempt.get("role") == "reviewer": + used += 1 + return used + return sum( + attempt.get("role") == "lead" + and attempt["created_at"] >= reviewer["created_at"] + for attempt in attempts + ) + + def main(argv=None): parser = argparse.ArgumentParser() parser.add_argument("--database", required=True); parser.add_argument("--artifacts", required=True) @@ -38,12 +69,31 @@ def main(argv=None): ) role = "lead" if headless_lead else "reviewer" workflow_role = ( - "internal_review_fixture" in snapshot or "review_adapter" in snapshot + "internal_review_fixture" in snapshot + or "review_adapter" in snapshot + or "review_adapters" in snapshot ) + profile_index = None + profile_id = None if workflow_role: - selected = snapshot["routing"]["roles"][role]["selected"]["profile"] + routed_role = snapshot["routing"]["roles"][role] + candidates = [routed_role["selected"], *routed_role["fallbacks"]] + profile_index = _profile_index( + store, args.run_id, role, handoff, + ) + if profile_index >= len(candidates): + raise ConflictError("frozen role fallback set is exhausted") + attempt_selection = candidates[profile_index] + profile_id = attempt_selection["profile_id"] + selected = attempt_selection["profile"] adapter_key = "lead_adapter" if headless_lead else "review_adapter" - adapter = snapshot.get(adapter_key) + adapters_key = "lead_adapters" if headless_lead else "review_adapters" + adapters = snapshot.get(adapters_key) + adapter = ( + adapters.get(profile_id) + if isinstance(adapters, dict) + else snapshot.get(adapter_key) if profile_index == 0 else None + ) identity = ExecutionIdentity( selected["harness"], adapter["harness_version"] if adapter else "fixture", @@ -62,7 +112,12 @@ def main(argv=None): else ("devsquad.codex_review_worker" if adapter else "devsquad.review_worker") ) command = [sys.executable, "-P", "-m", module] - worker_snapshot = dict(snapshot) + worker_snapshot = json.loads(canonical_json(snapshot)) + worker_snapshot["routing"]["roles"][role]["selected"] = attempt_selection + if adapter is None: + worker_snapshot.pop(adapter_key, None) + else: + worker_snapshot[adapter_key] = adapter if headless_lead: worker_snapshot["headless_handoff"] = { "handoff_id": handoff.handoff_id, @@ -71,7 +126,11 @@ def main(argv=None): } input_path, _, _ = store.finalize_artifact( args.run_id, - "lead-workflow-input.json" if headless_lead else "workflow-input.json", + ( + f"lead-workflow-input-{handoff.sequence}-{profile_index}.json" + if headless_lead + else f"workflow-input-{profile_index}.json" + ), canonical_json(worker_snapshot).encode(), ) stdin_path = str(input_path) @@ -98,7 +157,17 @@ def main(argv=None): environment, ) supervisor = Supervisor(store) - try: handle = supervisor.launch_durable(args.run_id, args.expected_version, spec, f"daemon:{os.getpid()}", args.package_digest, role=role if workflow_role else "worker") + try: + handle = supervisor.launch_durable( + args.run_id, + args.expected_version, + spec, + f"daemon:{os.getpid()}", + args.package_digest, + role=role if workflow_role else "worker", + profile_id=profile_id, + profile_index=profile_index, + ) except ConflictError: return 0 except BudgetExhausted: Service(Path(args.database).parent).fail_budget_exhausted( @@ -115,6 +184,11 @@ def main(argv=None): Service(Path(args.database).parent).resume(args.run_id) except ConflictError: pass + elif current["state"] == "queued" and current["phase"] is None: + try: + Service(Path(args.database).parent).resume(args.run_id) + except ConflictError: + pass return 0 if returncode == 0 else 1 if __name__ == "__main__": raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/lead_worker.py b/plugin/core/src/devsquad/lead_worker.py index 0793eac..bb9b914 100644 --- a/plugin/core/src/devsquad/lead_worker.py +++ b/plugin/core/src/devsquad/lead_worker.py @@ -18,6 +18,9 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: handoff = snapshot.get("headless_handoff") if not isinstance(fixture, dict) or not isinstance(handoff, dict): raise ContractError("offline headless lead snapshot is incomplete") + selected = snapshot["routing"]["roles"]["lead"]["selected"] + if selected.get("profile_id", "").endswith("-fixture-fail"): + raise ContractError("offline headless lead fixture requested a failed attempt") choice = { "schema_version": 1, "candidate_sha256": handoff["packet"]["candidate_sha256"], diff --git a/plugin/core/src/devsquad/migrations/008_attempt_routing.sql b/plugin/core/src/devsquad/migrations/008_attempt_routing.sql new file mode 100644 index 0000000..b1e332a --- /dev/null +++ b/plugin/core/src/devsquad/migrations/008_attempt_routing.sql @@ -0,0 +1,2 @@ +ALTER TABLE attempts ADD COLUMN profile_id TEXT; +ALTER TABLE attempts ADD COLUMN profile_index INTEGER; diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index 8a60a8e..be55931 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -241,9 +241,10 @@ def build_early_terminal_reports( "cancelled": bool(attempt.get("cancelled", state == "cancelled")), "timed_out": bool(attempt.get("timed_out", False)), "selected_profile": ( - frozen.get("routing", {}).get("roles", {}).get("reviewer", {}).get( - "selected" - ) + attempt.get("selected_profile") + or frozen.get("routing", {}).get("roles", {}).get( + attempt.get("role", "reviewer"), {} + ).get("selected") if isinstance(frozen.get("routing"), dict) else None ), "observed_identity": None, @@ -494,6 +495,7 @@ def build_terminal_reports( completed_at: str, error: dict[str, Any] | None = None, headless_leads: list[dict[str, Any]] | None = None, + failed_attempts: list[dict[str, Any]] | None = None, ) -> dict[str, bytes]: if not isinstance(run_id, str) or not run_id: raise ContractError("report run id is invalid") @@ -552,7 +554,19 @@ def build_terminal_reports( "it is not live-provider evidence." ) lead_attempts = [evidence["attempt"] for evidence in lead_evidence] - all_attempts = attempts + lead_attempts + failed = [] if failed_attempts is None else failed_attempts + if not isinstance(failed, list) or not all( + isinstance(attempt, dict) + and attempt.get("role") in {"reviewer", "lead"} + for attempt in failed): + raise ContractError("failed fallback attempts are invalid") + failed_reviewers = [ + attempt for attempt in failed if attempt["role"] == "reviewer" + ] + failed_leads = [attempt for attempt in failed if attempt["role"] == "lead"] + reviewer_attempts = failed_reviewers + attempts + lead_attempts = failed_leads + lead_attempts + all_attempts = reviewer_attempts + lead_attempts all_native_counts = [ attempt["native_model_requests"] for attempt in all_attempts ] @@ -572,7 +586,7 @@ def build_terminal_reports( "checks": final_packet["checks"], "evaluation": final_packet["evaluation"], "criteria": final_packet["evaluation"]["criteria"], - "attempts": attempts, + "attempts": reviewer_attempts, "dispositions": dispositions, "revisions": { "requested": sum( diff --git a/plugin/core/src/devsquad/review_worker.py b/plugin/core/src/devsquad/review_worker.py index f6cf36f..18d41b6 100644 --- a/plugin/core/src/devsquad/review_worker.py +++ b/plugin/core/src/devsquad/review_worker.py @@ -182,6 +182,9 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: fixture = snapshot.get("internal_review_fixture") if not isinstance(fixture, dict): raise ContractError("offline review snapshot is incomplete") + selected = snapshot["routing"]["roles"]["reviewer"]["selected"] + if selected.get("profile_id", "").endswith("-fixture-fail"): + raise ContractError("offline reviewer fixture requested a failed attempt") return run_review_and_checks(snapshot, fixture) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index e1666d8..e48019a 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -131,8 +131,17 @@ def _paused_review_terminal_artifacts( } artifacts = store.artifacts_for_run(run_id) artifacts_by_name = {artifact["name"]: artifact for artifact in artifacts} + failed_by_id = { + attempt["id"]: attempt + for attempt in self._failed_fallback_attempts( + store, run_id, snapshot, + ) + } attempts = [] for attempt in store.attempts_for_run(run_id): + if attempt["id"] in failed_by_id: + attempts.append(failed_by_id[attempt["id"]]) + continue if attempt.get("role") == "reviewer": packet = packets_by_attempt.get(attempt["id"]) if packet is not None: @@ -222,6 +231,9 @@ def fail_budget_exhausted( headless_leads=self._headless_leads_for_history( snapshot, history, run_artifacts, ), + failed_attempts=self._failed_fallback_attempts( + store, run_id, snapshot, + ), ) terminal_artifacts = [] for name in sorted(reports): @@ -504,14 +516,28 @@ def _continue_preparation( capacity_in_flight=store.active_pool_counts(), ) if internal_delay is None and internal_review_fixture is None: - snapshot["review_adapter"] = freeze_codex_reviewer( - snapshot["routing"]["roles"]["reviewer"]["selected"], - ) + reviewer_route = snapshot["routing"]["roles"]["reviewer"] + reviewer_candidates = [ + reviewer_route["selected"], *reviewer_route["fallbacks"], + ] + snapshot["review_adapters"] = { + candidate["profile_id"]: freeze_codex_reviewer(candidate) + for candidate in reviewer_candidates + } + snapshot["review_adapter"] = snapshot["review_adapters"][ + reviewer_route["selected"]["profile_id"] + ] if (internal_delay is None and task["lead"]["mode"] == "headless" and internal_lead_fixture is None): - snapshot["lead_adapter"] = freeze_codex_lead( - snapshot["routing"]["roles"]["lead"]["selected"], - ) + lead_route = snapshot["routing"]["roles"]["lead"] + lead_candidates = [lead_route["selected"], *lead_route["fallbacks"]] + snapshot["lead_adapters"] = { + candidate["profile_id"]: freeze_codex_lead(candidate) + for candidate in lead_candidates + } + snapshot["lead_adapter"] = snapshot["lead_adapters"][ + lead_route["selected"]["profile_id"] + ] if store.remaining_wall_seconds(run_id) == 0: raise BudgetExhausted("run wall-time budget is exhausted in preflight") package, digest = self._freeze_package() @@ -806,6 +832,51 @@ def _headless_leads_for_history( )) return evidence + @staticmethod + def _failed_fallback_attempts( + store: Store, + run_id: str, + snapshot: dict[str, Any], + ) -> list[dict[str, Any]]: + failures = [] + for attempt in store.attempts_for_run(run_id): + encoded = attempt.get("output_metadata") + if not encoded: + continue + try: + metadata = json.loads(encoded) + if not isinstance(metadata, dict) or "failure" not in metadata: + continue + error = metadata["failure"] + role = attempt["role"] + index = attempt["profile_index"] + routed = snapshot["routing"]["roles"][role] + candidates = [routed["selected"], *routed["fallbacks"]] + selected = candidates[index] + except (IndexError, KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError("saved fallback attempt is invalid") from exc + if (not isinstance(error, dict) + or selected["profile_id"] != attempt["profile_id"]): + raise ConflictError("saved fallback attempt changed its profile") + failures.append({ + "id": attempt["id"], + "role": role, + "status": "failed", + "profile_index": index, + "selected_profile": selected, + "observed_identity": None, + "worker_invocations": 1, + "native_model_requests": None, + "usage": { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + "error": error, + }) + return failures + def _terminalize_branch_review( self, store: Store, @@ -847,6 +918,9 @@ def _terminalize_branch_review( completed_at=datetime.now(timezone.utc).isoformat(), error=error, headless_leads=headless_leads, + failed_attempts=self._failed_fallback_attempts( + store, run_id, snapshot, + ), ) prepared = [] for name in sorted(reports): diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 8715b7e..a3d7d4a 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 7 +SUPPORTED_SCHEMA_VERSION = 8 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -674,14 +674,24 @@ def _enforce_attempt_budget( run_id: str, run: sqlite3.Row, ) -> None: + try: + snapshot = json.loads(run["mutable_snapshot"] or "null") + max_invocations = snapshot["task"]["budget"]["max_worker_invocations"] + except (KeyError, TypeError, json.JSONDecodeError): + max_invocations = None + if type(max_invocations) is int: + launched = self.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()[0] + if launched >= max_invocations: + raise BudgetExhausted("worker invocation budget is exhausted") wall_seconds = self._wall_seconds_from_run(run) - if wall_seconds is None: - return - elapsed_ms = self._execution_elapsed_ms( - run_id, run, _authoritative_now(), - ) - if wall_seconds * 1000 - elapsed_ms < 1000: - raise BudgetExhausted("run wall-time budget is exhausted") + if wall_seconds is not None: + elapsed_ms = self._execution_elapsed_ms( + run_id, run, _authoritative_now(), + ) + if wall_seconds * 1000 - elapsed_ms < 1000: + raise BudgetExhausted("run wall-time budget is exhausted") def active_pool_counts(self) -> dict[str, int]: return { @@ -703,6 +713,8 @@ def reserve_attempt( role: str = "worker", *, account_pool_id: str | None = None, + profile_id: str | None = None, + profile_index: int | None = None, ) -> AttemptReservation: if not owner_id or not package_digest: raise ContractError("supervisor owner and package digest are required") @@ -728,11 +740,26 @@ def reserve_attempt( raise ContractError("attempt account pool is invalid") try: snapshot = json.loads(run["mutable_snapshot"]) - selected = snapshot["routing"]["roles"][role]["selected"]["profile"] + routed_role = snapshot["routing"]["roles"][role] + candidates = [ + routed_role["selected"], *routed_role.get("fallbacks", []), + ] + if profile_id is None and profile_index is None: + selected = candidates[0] + elif (type(profile_index) is int + and 0 <= profile_index < len(candidates) + and profile_id == candidates[profile_index]["profile_id"]): + selected = candidates[profile_index] + else: + raise ConflictError( + "attempt profile does not match frozen routing order" + ) capacity = snapshot["routing"]["capacity"][account_pool_id] - expected_pool = selected["account_pool_id"] + expected_pool = selected["profile"]["account_pool_id"] max_concurrency = capacity["max_concurrency"] - except (KeyError, TypeError, json.JSONDecodeError) as exc: + except ConflictError: + raise + except (IndexError, KeyError, TypeError, json.JSONDecodeError) as exc: raise ConflictError( "frozen account-pool reservation is invalid" ) from exc @@ -749,16 +776,19 @@ def reserve_attempt( ).fetchone()[0] if in_flight >= max_concurrency: raise ConflictError("account pool concurrency is full") + elif profile_id is not None or profile_index is not None: + raise ContractError("attempt profile requires an account pool") old = self.connection.execute("SELECT COALESCE(MAX(fencing_token),0) FROM supervisor_claims WHERE run_id=?", (run_id,)).fetchone()[0] supervisor_token, attempt_id, attempt_token = old + 1, str(uuid.uuid4()), uuid.uuid4().hex now, version = _utc_now(), expected_version + 1 self.connection.execute("INSERT OR REPLACE INTO supervisor_claims(run_id,owner_id,fencing_token,package_digest,heartbeat_at,active) VALUES(?,?,?,?,?,1)", (run_id, owner_id, supervisor_token, package_digest, now)) self.connection.execute( "INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token," - "status,heartbeat_at,package_digest,created_at,role,account_pool_id) " - "VALUES(?,?,?,?,?,'reserved',?,?,?,?,?)", + "status,heartbeat_at,package_digest,created_at,role,account_pool_id," + "profile_id,profile_index) VALUES(?,?,?,?,?,'reserved',?,?,?,?,?,?,?)", (attempt_id, run_id, run["project_id"], run["worktree_path"], - attempt_token, now, package_digest, now, role, account_pool_id), + attempt_token, now, package_digest, now, role, account_pool_id, + profile_id, profile_index), ) self.connection.execute("UPDATE runs SET phase='launching',version=?,updated_at=? WHERE id=?", (version, now, run_id)) event = canonical_json({"attempt_id": attempt_id, "supervisor_token": supervisor_token}) @@ -1085,6 +1115,78 @@ def commit_durable_import(self, run_id: str, attempt_token: str, artifacts: list self.connection.execute("ROLLBACK") raise + def commit_durable_fallback( + self, + run_id: str, + attempt_token: str, + artifacts: list[dict[str, Any]], + metadata: Any, + error: dict[str, Any], + ) -> str: + """Record one failed profile attempt and queue its frozen fallback.""" + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=False, + ) + enriched_metadata = dict(metadata) + enriched_metadata["failure"] = json.loads(canonical_json(error)) + encoded_metadata = canonical_json(enriched_metadata) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,stdout_artifact_id,stderr_artifact_id,output_metadata " + "FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("durable fallback import is fenced") + if (attempt["status"] == "finished" and run["state"] == "queued" + and run["phase"] is None): + self.connection.execute("COMMIT") + return "queued" + if (attempt["status"] != "running" or run["state"] != "running" + or run["phase"] is not None): + raise ConflictError("durable fallback import is fenced") + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET state='queued',phase=NULL,version=?,updated_at=? " + "WHERE id=?", + (version, now, run_id), + ) + payload = dict(error) + payload["attempt_id"] = attempt["id"] + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'run.fallback_queued',?,?)", + (run_id, version, canonical_json(payload), now), + ) + self.connection.execute("COMMIT") + return "queued" + except Exception: + self.connection.execute("ROLLBACK") + raise + def commit_durable_handoff( self, run_id: str, diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 74598f0..05271cf 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -165,6 +165,50 @@ def __init__(self, store: Store, *, output_limit: int = 1024 * 1024, grace_secon raise ContractError("supervisor bounds must be positive") self.store, self.output_limit, self.grace_seconds = store, output_limit, grace_seconds + def _failed_fallback_attempts( + self, + run_id: str, + snapshot: dict[str, Any], + current_attempt_id: str, + ) -> list[dict[str, Any]]: + failures = [] + for attempt in self.store.attempts_for_run(run_id): + if attempt["id"] == current_attempt_id or not attempt.get("output_metadata"): + continue + try: + metadata = json.loads(attempt["output_metadata"]) + if not isinstance(metadata, dict) or "failure" not in metadata: + continue + error = metadata["failure"] + role = attempt["role"] + index = attempt["profile_index"] + routed = snapshot["routing"]["roles"][role] + candidates = [routed["selected"], *routed["fallbacks"]] + selected = candidates[index] + except (IndexError, KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError("saved fallback attempt is invalid") from exc + if (not isinstance(error, dict) + or selected["profile_id"] != attempt["profile_id"]): + raise ConflictError("saved fallback attempt changed its profile") + failures.append({ + "id": attempt["id"], + "role": role, + "status": "failed", + "profile_index": index, + "selected_profile": selected, + "observed_identity": None, + "worker_invocations": 1, + "native_model_requests": None, + "usage": { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + "error": error, + }) + return failures + def launch(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: str, package_digest: str) -> RunningAttempt: reservation = self.store.reserve_attempt( run_id, @@ -224,6 +268,8 @@ def launch_durable( package_digest: str, *, role: str = "worker", + profile_id: str | None = None, + profile_index: int | None = None, ) -> DurableAttempt: reservation = self.store.reserve_attempt( run_id, @@ -232,6 +278,8 @@ def launch_durable( package_digest, role, account_pool_id=spec.requested.account_pool, + profile_id=profile_id, + profile_index=profile_index, ) directory = self.store.artifacts / run_id / f".{reservation.attempt_id}.spool" directory.mkdir(parents=True, exist_ok=False) @@ -428,7 +476,9 @@ def import_durable(self, run_id: str) -> str: artifacts.append({"name":logical,"path":path,"sha256":digest,"byte_size":size}) snapshot=json.loads(self.store.run(run_id)["mutable_snapshot"]) workflow_review = ( - "internal_review_fixture" in snapshot or "review_adapter" in snapshot + "internal_review_fixture" in snapshot + or "review_adapter" in snapshot + or "review_adapters" in snapshot ) role = attempt.get("role", "worker") semantic_error=None @@ -479,7 +529,9 @@ def import_durable(self, run_id: str) -> str: ) payload["message"]=semantic_error if workflow_review: - prior_attempts = [] + prior_attempts = self._failed_fallback_attempts( + run_id, snapshot, attempt["id"], + ) if role == "lead": handoff = self.store.handoff_snapshot(run_id) if handoff is not None and isinstance(handoff.packet, dict): @@ -522,6 +574,42 @@ def import_durable(self, run_id: str) -> str: "returncode": receipt["returncode"], } payload.update(report_error) + candidates = [] + profile_index = None + try: + routed_role = snapshot["routing"]["roles"][role] + candidates = [ + routed_role["selected"], *routed_role["fallbacks"], + ] + profile_index = attempt.get("profile_index") + next_profile = candidates[profile_index + 1] + can_fallback = ( + terminal == "failed" + and type(profile_index) is int + and profile_index >= 0 + and profile_index + 1 < len(candidates) + and len(self.store.attempts_for_run(run_id)) + < snapshot["task"]["budget"]["max_worker_invocations"] + and self.store.remaining_wall_seconds(run_id) not in {0} + ) + except (IndexError, KeyError, TypeError): + can_fallback = False + next_profile = None + if can_fallback: + fallback_error = { + **(report_error or {}), + "role": role, + "failed_profile_id": attempt.get("profile_id"), + "failed_profile_index": profile_index, + "next_profile_id": next_profile["profile_id"], + } + return self.store.commit_durable_fallback( + run_id, + attempt["attempt_token"], + artifacts, + metadata, + fallback_error, + ) reports = build_early_terminal_reports( run_id=run_id, state=terminal, @@ -529,7 +617,6 @@ def import_durable(self, run_id: str) -> str: snapshot=snapshot, run_artifacts=( self.store.artifacts_for_run(run_id) + artifacts - if role == "lead" else artifacts ), events=self.store.events_for_run(run_id), completed_at=datetime.now(timezone.utc).isoformat(), @@ -541,6 +628,12 @@ def import_durable(self, run_id: str) -> str: "returncode": receipt["returncode"], "cancelled": receipt["cancelled"], "timed_out": receipt["timed_out"], + "selected_profile": ( + candidates[profile_index] + if type(profile_index) is int + and 0 <= profile_index < len(candidates) + else None + ), }, prior_attempts=prior_attempts, ) diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index 6b67c98..d03f084 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -468,6 +468,40 @@ def build_review_prompt(task: dict[str, Any], workspace: dict[str, Any]) -> str: ]) +def _frozen_attempt_selection( + snapshot: dict[str, Any], + role: str, + actual: Any, +) -> dict[str, Any]: + try: + routed = snapshot["routing"]["roles"][role] + candidates = [routed["selected"], *routed.get("fallbacks", [])] + except (KeyError, TypeError) as exc: + raise ContractError(f"frozen {role} selection is missing") from exc + for candidate in candidates: + if canonical_json(actual) == canonical_json(candidate): + return candidate + raise ContractError(f"{role} attempt is outside the frozen fallback set") + + +def _frozen_role_adapter( + snapshot: dict[str, Any], + role: str, + selection: dict[str, Any], +) -> Any: + plural_key = "review_adapters" if role == "reviewer" else "lead_adapters" + singular_key = "review_adapter" if role == "reviewer" else "lead_adapter" + adapters = snapshot.get(plural_key) + if adapters is not None: + if not isinstance(adapters, dict): + raise ContractError(f"frozen {role} adapters are invalid") + adapter = adapters.get(selection["profile_id"]) + if adapter is None: + raise ContractError(f"frozen {role} fallback adapter is missing") + return adapter + return snapshot.get(singular_key) + + def validate_headless_lead_choice( value: dict[str, Any], packet: dict[str, Any], @@ -574,17 +608,14 @@ def validate_headless_lead_evidence( }, "headless lead attempt evidence") if attempt["role"] != "lead": raise ContractError("headless lead attempt role is invalid") - try: - frozen_lead = snapshot["routing"]["roles"]["lead"]["selected"] - except (KeyError, TypeError) as exc: - raise ContractError("frozen lead selection is missing") from exc - if canonical_json(attempt["selected_profile"]) != canonical_json(frozen_lead): - raise ContractError("headless lead attempt changes the selected profile") + frozen_lead = _frozen_attempt_selection( + snapshot, "lead", attempt["selected_profile"], + ) prompt_sha256 = hashlib.sha256(build_lead_prompt(task, packet).encode()).hexdigest() if _sha256(attempt["prompt_sha256"], "headless lead prompt sha256") != prompt_sha256: raise ContractError("headless lead prompt hash does not match the handoff") - adapter = snapshot.get("lead_adapter") + adapter = _frozen_role_adapter(snapshot, "lead", frozen_lead) observed = attempt["observed_identity"] native_ids = attempt["native_ids"] if adapter is None: @@ -726,19 +757,16 @@ def validate_branch_review_evidence( }, "review attempt evidence") if attempt["role"] != "reviewer": raise ContractError("review attempt role is invalid") - try: - frozen_reviewer = snapshot["routing"]["roles"]["reviewer"]["selected"] - except (KeyError, TypeError) as exc: - raise ContractError("frozen reviewer selection is missing") from exc - if canonical_json(attempt["selected_profile"]) != canonical_json(frozen_reviewer): - raise ContractError("review attempt changes the frozen selected profile") + frozen_reviewer = _frozen_attempt_selection( + snapshot, "reviewer", attempt["selected_profile"], + ) prompt_sha256 = hashlib.sha256(build_review_prompt(task, workspace).encode()).hexdigest() if _sha256(attempt["prompt_sha256"], "review prompt_sha256") != prompt_sha256: raise ContractError("review prompt hash does not match the frozen prompt") review_sha256 = hashlib.sha256(canonical_json(review).encode()).hexdigest() if _sha256(attempt["review_sha256"], "review document sha256") != review_sha256: raise ContractError("review document hash is invalid") - adapter = snapshot.get("review_adapter") + adapter = _frozen_role_adapter(snapshot, "reviewer", frozen_reviewer) observed = attempt["observed_identity"] native_ids = attempt["native_ids"] if adapter is None: diff --git a/test/core/test_cli.py b/test/core/test_cli.py index a53d7bf..92ea72c 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -291,7 +291,7 @@ def build_python(): return candidate return None - def test_installed_wheel_contains_and_applies_migrations_through_seven(self): + def test_installed_wheel_contains_and_applies_migrations_through_eight(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") @@ -333,9 +333,9 @@ def test_installed_wheel_contains_and_applies_migrations_through_seven(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 7 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 8 attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} - assert {"role", "account_pool_id"} <= attempt_columns + assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} assert {"package_path", "package_digest", "supersedes_run_id"} <= columns assert store.connection.execute( diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 3a86559..b15e908 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 7, + 8, ) tables = { row[0] @@ -618,7 +618,7 @@ def build_python(): return candidate return None - def test_installed_wheel_applies_schema_four_to_seven(self): + def test_installed_wheel_applies_schema_four_to_eight(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel") @@ -684,12 +684,14 @@ def test_installed_wheel_applies_schema_four_to_seven(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 7 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 8 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() claim_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(claims)")} assert {"handoff_id", "lease_expires_at", "renewed_at"} <= claim_columns + attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} + assert {"profile_id", "profile_index"} <= attempt_columns finally: store.close() ''' diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index e3903b5..8fcd384 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -164,6 +164,79 @@ def configure_fixture_headless(self): self.task["lead"] = {"mode": "headless"} self.task["budget"]["max_worker_invocations"] = 2 + def configure_reviewer_fallback(self): + profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) + failing = profiles["profiles"][0] + failing["id"] = "reviewer-fixture-fail" + failing["model_id"] = "fixture-model-fail" + fallback = dict(failing) + fallback.update({ + "id": "reviewer-fallback", + "model_family": "fixture-family-b", + "model_id": "fixture-model-fallback", + }) + profiles["profiles"].append(fallback) + profiles["bindings"]["review.deep"] = { + "profile_id": failing["id"], "version": 2, + } + policy = json.loads((self.repo / "devsquad/policy.json").read_text()) + policy["roles"]["reviewer"] = [ + {"kind": "alias", "id": "review.deep"}, + {"kind": "profile", "id": fallback["id"]}, + ] + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "fallback routing"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["budget"]["max_fallbacks_per_step"] = 1 + + def configure_headless_lead_fallback(self): + self.configure_fixture_headless() + profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) + failing = next(profile for profile in profiles["profiles"] + if profile["id"] == "fixture-lead") + failing.update({ + "id": "lead-fixture-fail", + "model_id": "fixture-lead-model-fail", + }) + fallback = dict(failing) + fallback.update({ + "id": "lead-fallback", + "model_family": "fixture-family-lead-fallback", + "model_id": "fixture-lead-model-fallback", + }) + profiles["profiles"].append(fallback) + profiles["bindings"]["lead.primary"] = { + "profile_id": failing["id"], "version": 2, + } + policy = json.loads((self.repo / "devsquad/policy.json").read_text()) + policy["roles"]["lead"] = [ + {"kind": "alias", "id": "lead.primary"}, + {"kind": "profile", "id": fallback["id"]}, + ] + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "lead fallback routing"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["budget"]["max_worker_invocations"] = 3 + self.task["budget"]["max_fallbacks_per_step"] = 1 + def test_detached_review_imports_bound_evidence_and_publishes_host_handoff(self): (self.repo / "notes.txt").write_text("unrelated local work\n") before_head = self.git_bytes("rev-parse", "HEAD") @@ -788,6 +861,116 @@ def test_public_preflight_observes_live_shared_pool_reservations(self): <= {artifact["name"] for artifact in result["artifacts"]} ) + def test_failed_reviewer_uses_one_frozen_fallback_and_keeps_both_attempts(self): + self.configure_reviewer_fallback() + self.task["budget"]["max_worker_invocations"] = 2 + run_id, waiting = self.start_waiting("reviewer-fallback") + claimed = self.service.handoff_claim( + run_id, waiting["version"], "host-fallback", + ) + packet = claimed["handoff"]["packet"] + self.assertEqual( + packet["attempt"]["selected_profile"]["profile_id"], + "reviewer-fallback", + ) + accepted = self.decision( + packet, "accept-fallback", "accept", "Fallback review accepted.", + ) + completed = self.service.handoff_complete( + run_id, claimed["claim"], accepted, + ) + self.assertEqual(completed["state"], "succeeded") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(run_id)["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual( + [attempt["selected_profile"]["profile_id"] + for attempt in receipt["attempts"]], + ["reviewer-fixture-fail", "reviewer-fallback"], + ) + self.assertEqual(receipt["attempts"][0]["status"], "failed") + self.assertEqual(receipt["attempts"][1]["role"], "reviewer") + + def test_fallback_none_does_not_retry_a_failed_reviewer(self): + self.configure_reviewer_fallback() + self.task["routing"]["overrides"] = { + "reviewer": { + "profile_id": "reviewer-fixture-fail", + "fallback": "none", + }, + } + self.task["budget"]["max_worker_invocations"] = 3 + started = self.service.start( + self.task, "reviewer-no-fallback", _internal_review_fixture=self.fixture, + ) + completed = self.wait_state(started["run_id"], {"failed"}) + self.assertEqual(completed["state"], "failed") + result = self.service.result(started["run_id"]) + artifacts = {artifact["name"]: artifact for artifact in result["artifacts"]} + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["routing"]["roles"]["reviewer"]["fallbacks"], []) + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + self.assertEqual( + receipt["attempts"][0]["selected_profile"]["profile_id"], + "reviewer-fixture-fail", + ) + self.assertNotIn( + "run.fallback_queued", + {event["type"] for event in self.service.events(started["run_id"])["events"]}, + ) + + def test_worker_invocation_budget_blocks_a_frozen_reviewer_fallback(self): + self.configure_reviewer_fallback() + self.task["budget"]["max_worker_invocations"] = 1 + started = self.service.start( + self.task, "reviewer-budget-no-fallback", + _internal_review_fixture=self.fixture, + ) + completed = self.wait_state(started["run_id"], {"failed"}) + self.assertEqual(completed["state"], "failed") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(started["run_id"])["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual( + [item["profile_id"] for item in + receipt["routing"]["roles"]["reviewer"]["fallbacks"]], + ["reviewer-fallback"], + ) + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + + def test_failed_headless_lead_uses_its_own_frozen_fallback(self): + self.configure_headless_lead_fallback() + started = self.service.start( + self.task, + "headless-lead-fallback", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "accept", + "reason": "The frozen review evidence is sufficient.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "succeeded") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(started["run_id"])["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["accounting"]["worker_invocations"], 3) + self.assertEqual( + [attempt["selected_profile"]["profile_id"] + for attempt in receipt["lead"]["attempts"]], + ["lead-fixture-fail", "lead-fallback"], + ) + self.assertEqual(receipt["lead"]["attempts"][0]["status"], "failed") + def test_headless_lead_is_a_second_fenced_attempt_and_terminalizes_automatically(self): self.configure_fixture_headless() started = self.service.start( diff --git a/test/core/test_review_workflow.py b/test/core/test_review_workflow.py index a7cc31b..a8bce82 100644 --- a/test/core/test_review_workflow.py +++ b/test/core/test_review_workflow.py @@ -291,7 +291,7 @@ def test_combined_evidence_recomputes_gates_profile_and_accounting(self): "derived gates"), (lambda value: value["attempt"].__setitem__( "selected_profile", {"profile_id": "substituted"}), - "frozen selected profile"), + "frozen fallback set"), (lambda value: value["attempt"].__setitem__("worker_invocations", 2), "invocation accounting"), (lambda value: value["attempt"]["usage"].__setitem__("total_tokens", 0), diff --git a/test/core/test_store.py b/test/core/test_store.py index 6d69664..354807a 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -354,8 +354,8 @@ def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): ) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 7) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(8,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 8) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(9,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -370,13 +370,15 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 7) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 8) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") } self.assertIn("role", attempt_columns) self.assertIn("account_pool_id", attempt_columns) + self.assertIn("profile_id", attempt_columns) + self.assertIn("profile_index", attempt_columns) def test_version_three_fixture_adds_run_snapshot_columns(self): old_db=self.root/"v3.sqlite3"; connection=sqlite3.connect(old_db) @@ -385,7 +387,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],7) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],8) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From 1737667a3a847bfd09559abd5a53f8ec88745982 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 22 Sep 2026 20:40:11 +0530 Subject: [PATCH 065/197] fix: close M3 audit findings --- plugin/core/src/devsquad/detached.py | 27 ++++- plugin/core/src/devsquad/reports.py | 17 ++- plugin/core/src/devsquad/service.py | 68 +++++++++--- plugin/core/src/devsquad/store.py | 38 +++++-- plugin/core/src/devsquad/supervisor.py | 68 ++---------- test/core/test_review_runtime.py | 147 +++++++++++++++++++++++++ test/core/test_store.py | 62 ++++++++++- 7 files changed, 340 insertions(+), 87 deletions(-) diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index 9beaa89..42792ad 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -11,6 +11,24 @@ from .supervisor import Supervisor +def _is_saved_fallback_failure(attempt) -> bool: + encoded = attempt.get("output_metadata") + if not encoded: + return False + try: + metadata = json.loads(encoded) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("saved attempt metadata is invalid") from exc + if not isinstance(metadata, dict): + raise ConflictError("saved attempt metadata is invalid") + failure = metadata.get("failure") + if failure is None: + return False + if not isinstance(failure, dict): + raise ConflictError("saved fallback failure is invalid") + return True + + def _profile_index( store: Store, run_id: str, @@ -19,7 +37,10 @@ def _profile_index( ) -> int: attempts = store.attempts_for_run(run_id) if handoff is None: - return sum(attempt.get("role") == role for attempt in attempts) + return sum( + attempt.get("role") == role and _is_saved_fallback_failure(attempt) + for attempt in attempts + ) reviewer_id = handoff.packet.get("attempt_id") reviewer = next( (attempt for attempt in attempts if attempt["id"] == reviewer_id), None, @@ -32,12 +53,14 @@ def _profile_index( for attempt in attempts: if attempt["id"] == reviewer_id: seen = True - elif seen and attempt.get("role") == "reviewer": + elif (seen and attempt.get("role") == "reviewer" + and _is_saved_fallback_failure(attempt)): used += 1 return used return sum( attempt.get("role") == "lead" and attempt["created_at"] >= reviewer["created_at"] + and _is_saved_fallback_failure(attempt) for attempt in attempts ) diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index be55931..df465ba 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -203,6 +203,7 @@ def build_early_terminal_reports( error: dict[str, Any] | None, attempt: dict[str, Any] | None = None, prior_attempts: list[dict[str, Any]] | None = None, + prior_dispositions: list[dict[str, Any]] | None = None, ) -> dict[str, bytes]: """Build the M3 report set when no valid handoff/lead decision exists.""" if not isinstance(run_id, str) or not run_id: @@ -230,6 +231,10 @@ def build_early_terminal_reports( if not isinstance(prior, list) or not all( isinstance(item, dict) for item in prior): raise ContractError("early terminal prior attempts are invalid") + dispositions = [] if prior_dispositions is None else prior_dispositions + if not isinstance(dispositions, list) or not all( + isinstance(item, dict) for item in dispositions): + raise ContractError("early terminal prior dispositions are invalid") if attempt is not None: if not isinstance(attempt, dict) or not isinstance(attempt.get("id"), str): raise ContractError("early terminal report attempt is invalid") @@ -284,7 +289,7 @@ def build_early_terminal_reports( "evaluation": None, "criteria": criteria, "attempts": prior + ([attempt_projection] if attempt_projection else []), - "dispositions": [], + "dispositions": dispositions, "lead": { "mode": task.get("lead", {}).get("mode"), "status": ( @@ -432,6 +437,16 @@ def _history( return attempts, dispositions +def project_branch_review_history( + entries: list[dict[str, Any]], + snapshot: dict[str, Any], +) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + """Validate and project any completed review revisions for early reports.""" + if entries == []: + return [], [] + return _history(entries, snapshot) + + def _markdown(receipt: dict[str, Any]) -> str: review = receipt["review"] lines = [ diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index e48019a..54fea98 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -21,7 +21,11 @@ ContractError, ProfileUnsupported, ) -from .reports import build_early_terminal_reports, build_terminal_reports +from .reports import ( + build_early_terminal_reports, + build_terminal_reports, + project_branch_review_history, +) from .router import capacity_with_live_reservations, load_routing from .store import ( ConflictError, @@ -93,19 +97,16 @@ def _preparation_failure_artifacts( }) return prepared - def _paused_review_terminal_artifacts( - self, + @staticmethod + def _saved_review_progress( store: Store, run_id: str, snapshot: dict[str, Any], handoff: HandoffSnapshot | None, - *, - state: str, - phase: str, - error: dict[str, Any] | None, - ) -> list[dict[str, Any]]: - """Materialize complete M3 reports before a paused run terminalizes.""" + ) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + """Project every persisted attempt and completed disposition in DB order.""" history = store.branch_review_history(run_id) + _, dispositions = project_branch_review_history(history, snapshot) frozen_handoffs = { item["handoff_id"]: { "handoff_id": item["handoff_id"], @@ -133,7 +134,7 @@ def _paused_review_terminal_artifacts( artifacts_by_name = {artifact["name"]: artifact for artifact in artifacts} failed_by_id = { attempt["id"]: attempt - for attempt in self._failed_fallback_attempts( + for attempt in Service._failed_fallback_attempts( store, run_id, snapshot, ) } @@ -175,6 +176,24 @@ def _paused_review_terminal_artifacts( "status": "succeeded", **evidence["attempt"], }) + return attempts, dispositions + + def _paused_review_terminal_artifacts( + self, + store: Store, + run_id: str, + snapshot: dict[str, Any], + handoff: HandoffSnapshot | None, + *, + state: str, + phase: str, + error: dict[str, Any] | None, + ) -> list[dict[str, Any]]: + """Materialize complete M3 reports before a paused run terminalizes.""" + attempts, dispositions = self._saved_review_progress( + store, run_id, snapshot, handoff, + ) + artifacts = store.artifacts_for_run(run_id) reports = build_early_terminal_reports( run_id=run_id, state=state, @@ -186,6 +205,7 @@ def _paused_review_terminal_artifacts( phase=phase, error=error, prior_attempts=attempts, + prior_dispositions=dispositions, ) prepared = [] for name in sorted(reports): @@ -1085,10 +1105,32 @@ def _continue_headless_lead( ) if saved is None: queued = store.queue_headless_lead(run_id, run["version"]) - if queued["action"] != "queued": - raise ConflictError( - "headless lead worker invocation budget is exhausted" + if queued["action"] == "budget_exhausted": + error = { + "error": "BUDGET_EXHAUSTED", + "message": "headless lead worker invocation budget is exhausted", + } + terminal_artifacts = self._paused_review_terminal_artifacts( + store, + run_id, + snapshot, + handoff, + state="failed", + phase="lead", + error=error, ) + version = store.fail_queued_budget( + run_id, run["version"], error, terminal_artifacts, + ) + return { + "action": "budget_exhausted", + "state": "failed", + "version": version, + "replayed_continuation": False, + "launch": None, + } + if queued["action"] != "queued": + raise ConflictError("headless lead handoff was not queueable") prepared = store.run(run_id) package, digest = self._verified_package(prepared) return { diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index a3d7d4a..9f0721d 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -680,9 +680,7 @@ def _enforce_attempt_budget( except (KeyError, TypeError, json.JSONDecodeError): max_invocations = None if type(max_invocations) is int: - launched = self.connection.execute( - "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), - ).fetchone()[0] + launched = self.worker_invocations(run_id) if launched >= max_invocations: raise BudgetExhausted("worker invocation budget is exhausted") wall_seconds = self._wall_seconds_from_run(run) @@ -704,6 +702,13 @@ def active_pool_counts(self) -> dict[str, int]: ) } + def worker_invocations(self, run_id: str) -> int: + """Count attempts whose durable runner actually crossed the launch fence.""" + return self.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=? AND pid IS NOT NULL", + (run_id,), + ).fetchone()[0] + def reserve_attempt( self, run_id: str, @@ -757,6 +762,10 @@ def reserve_attempt( capacity = snapshot["routing"]["capacity"][account_pool_id] expected_pool = selected["profile"]["account_pool_id"] max_concurrency = capacity["max_concurrency"] + capacity_status = capacity.get("status", "available") + unknown_policy = capacity.get( + "unknown_capacity_policy", "allow_bounded", + ) except ConflictError: raise except (IndexError, KeyError, TypeError, json.JSONDecodeError) as exc: @@ -765,16 +774,28 @@ def reserve_attempt( ) from exc if (expected_pool != account_pool_id or type(max_concurrency) is not int - or max_concurrency < 1): + or max_concurrency < 1 + or capacity_status not in { + "available", "exhausted", "unknown", + } + or unknown_policy not in {"allow_bounded", "block"}): raise ConflictError( "attempt account pool does not match frozen routing" ) + if capacity_status == "exhausted": + raise ConflictError("account pool capacity is exhausted") + if (capacity_status == "unknown" + and unknown_policy == "block"): + raise ConflictError("unknown account pool capacity is blocked") + effective_concurrency = ( + 1 if capacity_status == "unknown" else max_concurrency + ) in_flight = self.connection.execute( "SELECT COUNT(*) FROM attempts WHERE account_pool_id=? " "AND status IN ('reserved','running','cancelling','ownership_ambiguous')", (account_pool_id,), ).fetchone()[0] - if in_flight >= max_concurrency: + if in_flight >= effective_concurrency: raise ConflictError("account pool concurrency is full") elif profile_id is not None or profile_index is not None: raise ContractError("attempt profile requires an account pool") @@ -1358,9 +1379,7 @@ def queue_headless_lead( ).fetchone() if active: raise ConflictError("headless lead already has an active supervisor") - invocations = self.connection.execute( - "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), - ).fetchone()[0] + invocations = self.worker_invocations(run_id) if invocations >= budget: self.connection.execute("COMMIT") return { @@ -2598,7 +2617,8 @@ def fail_queued_budget( run = self.connection.execute( "SELECT state,phase,version FROM runs WHERE id=?", (run_id,), ).fetchone() - if (not run or run["state"] != "queued" or run["phase"] is not None + if (not run or run["state"] not in {"queued", "awaiting_host"} + or run["phase"] is not None or run["version"] != expected_version): raise ConflictError("budget exhaustion is no longer current") if self.connection.execute( diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 05271cf..158be7b 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -165,50 +165,6 @@ def __init__(self, store: Store, *, output_limit: int = 1024 * 1024, grace_secon raise ContractError("supervisor bounds must be positive") self.store, self.output_limit, self.grace_seconds = store, output_limit, grace_seconds - def _failed_fallback_attempts( - self, - run_id: str, - snapshot: dict[str, Any], - current_attempt_id: str, - ) -> list[dict[str, Any]]: - failures = [] - for attempt in self.store.attempts_for_run(run_id): - if attempt["id"] == current_attempt_id or not attempt.get("output_metadata"): - continue - try: - metadata = json.loads(attempt["output_metadata"]) - if not isinstance(metadata, dict) or "failure" not in metadata: - continue - error = metadata["failure"] - role = attempt["role"] - index = attempt["profile_index"] - routed = snapshot["routing"]["roles"][role] - candidates = [routed["selected"], *routed["fallbacks"]] - selected = candidates[index] - except (IndexError, KeyError, TypeError, json.JSONDecodeError) as exc: - raise ConflictError("saved fallback attempt is invalid") from exc - if (not isinstance(error, dict) - or selected["profile_id"] != attempt["profile_id"]): - raise ConflictError("saved fallback attempt changed its profile") - failures.append({ - "id": attempt["id"], - "role": role, - "status": "failed", - "profile_index": index, - "selected_profile": selected, - "observed_identity": None, - "worker_invocations": 1, - "native_model_requests": None, - "usage": { - "input_tokens": None, - "output_tokens": None, - "total_tokens": None, - "source": "unavailable", - }, - "error": error, - }) - return failures - def launch(self, run_id: str, expected_version: int, spec: LaunchSpec, owner_id: str, package_digest: str) -> RunningAttempt: reservation = self.store.reserve_attempt( run_id, @@ -529,22 +485,13 @@ def import_durable(self, run_id: str) -> str: ) payload["message"]=semantic_error if workflow_review: - prior_attempts = self._failed_fallback_attempts( - run_id, snapshot, attempt["id"], + from .service import Service + prior_attempts, prior_dispositions = Service._saved_review_progress( + self.store, + run_id, + snapshot, + self.store.handoff_snapshot(run_id), ) - if role == "lead": - handoff = self.store.handoff_snapshot(run_id) - if handoff is not None and isinstance(handoff.packet, dict): - prior = handoff.packet.get("attempt") - if isinstance(prior, dict): - prior_attempts.append({ - "id": handoff.packet.get("attempt_id"), - "status": "succeeded", - **prior, - "review": handoff.packet.get("review"), - "checks": handoff.packet.get("checks"), - "evaluation": handoff.packet.get("evaluation"), - }) if receipt["cancelled"]: report_error = None elif receipt["timed_out"]: @@ -588,7 +535,7 @@ def import_durable(self, run_id: str) -> str: and type(profile_index) is int and profile_index >= 0 and profile_index + 1 < len(candidates) - and len(self.store.attempts_for_run(run_id)) + and self.store.worker_invocations(run_id) < snapshot["task"]["budget"]["max_worker_invocations"] and self.store.remaining_wall_seconds(run_id) not in {0} ) @@ -636,6 +583,7 @@ def import_durable(self, run_id: str) -> str: ), }, prior_attempts=prior_attempts, + prior_dispositions=prior_dispositions, ) for name in sorted(reports): content = reports[name] diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index 8fcd384..62e2e54 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -894,6 +894,54 @@ def test_failed_reviewer_uses_one_frozen_fallback_and_keeps_both_attempts(self): self.assertEqual(receipt["attempts"][0]["status"], "failed") self.assertEqual(receipt["attempts"][1]["role"], "reviewer") + def test_recovered_prelaunch_reservation_reuses_the_same_profile_and_budget(self): + self.task["budget"]["max_worker_invocations"] = 1 + with patch.object(self.service, "_spawn_daemon"): + started = self.service.start( + self.task, + "recover-review-reservation", + _internal_review_fixture=self.fixture, + ) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + run = store.run(started["run_id"]) + snapshot = json.loads(run["mutable_snapshot"]) + selected = snapshot["routing"]["roles"]["reviewer"]["selected"] + abandoned = store.reserve_attempt( + started["run_id"], + run["version"], + "abandoned-review-supervisor", + run["package_digest"], + "reviewer", + account_pool_id=selected["profile"]["account_pool_id"], + profile_id=selected["profile_id"], + profile_index=0, + ) + recovered_version = store.recover_launching( + started["run_id"], abandoned.version, + ) + self.assertEqual(store.worker_invocations(started["run_id"]), 0) + finally: + store.close() + + resumed = self.service.resume(started["run_id"]) + self.assertEqual(resumed["disposition"], "continued") + self.assertTrue(resumed["launched"]) + self.assertGreater(recovered_version, abandoned.version) + waiting = self.wait_state(started["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual( + [(attempt["status"], attempt["profile_index"]) + for attempt in attempts], + [("recovery_required", 0), ("finished", 0)], + ) + self.assertEqual(store.worker_invocations(started["run_id"]), 1) + finally: + store.close() + def test_fallback_none_does_not_retry_a_failed_reviewer(self): self.configure_reviewer_fallback() self.task["routing"]["overrides"] = { @@ -971,6 +1019,105 @@ def test_failed_headless_lead_uses_its_own_frozen_fallback(self): ) self.assertEqual(receipt["lead"]["attempts"][0]["status"], "failed") + def test_reviewer_fallback_exhausts_headless_lead_budget_terminally(self): + self.configure_reviewer_fallback() + profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) + lead = dict(profiles["profiles"][-1]) + lead.update({ + "id": "budgeted-headless-lead", + "model_family": "fixture-lead-family", + "model_id": "fixture-lead-model", + }) + profiles["profiles"].append(lead) + policy = json.loads((self.repo / "devsquad/policy.json").read_text()) + policy["roles"]["lead"] = [ + {"kind": "profile", "id": lead["id"]}, + ] + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) + subprocess.run(["git", "-C", str(self.repo), "add", "devsquad"], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "budgeted lead"], + check=True, + ) + self.task["project"]["target_ref"] = self.git_text("rev-parse", "HEAD").strip() + self.task["lead"] = {"mode": "headless"} + self.task["budget"]["max_worker_invocations"] = 2 + started = self.service.start( + self.task, + "reviewer-fallback-exhausts-lead", + _internal_review_fixture=self.fixture, + _internal_lead_fixture={ + "disposition": "accept", + "reason": "This lead must not launch beyond the budget.", + }, + ) + completed = self.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "failed") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(started["run_id"])["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(receipt["phase"], "lead") + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual(receipt["lead"]["status"], "failed") + + def test_cancelled_later_revision_keeps_prior_attempt_and_disposition(self): + marker = self.root / "slow-second-review" + self.task["budget"]["max_revisions"] = 1 + self.task["budget"]["max_worker_invocations"] = 2 + self.task["checks"] = [{ + "id": "slow-second-check", + "argv": [ + sys.executable, + "-c", + "from pathlib import Path; import sys,time; p=Path(sys.argv[1]); " + "time.sleep(30) if p.exists() else p.write_text('first')", + str(marker), + ], + "cwd": ".", + "timeout_seconds": 40, + "required_to_pass": False, + }] + run_id, first_wait = self.start_waiting("cancel-second-review") + first_claim = self.service.handoff_claim( + run_id, first_wait["version"], "host-first-revision", + ) + revise = self.decision( + first_claim["handoff"]["packet"], + "revise-before-cancel", + "revise", + "Run the frozen review one more time.", + ) + requeued = self.service.handoff_complete( + run_id, first_claim["claim"], revise, + ) + self.assertTrue(requeued["launched"]) + self.wait_state(run_id, {"running"}) + cancelled = self.service.cancel(run_id) + self.assertIn(cancelled["state"], {"cancelling", "cancelled"}) + self.assertEqual(self.wait_state(run_id, {"cancelled"})["state"], "cancelled") + artifacts = { + artifact["name"]: artifact + for artifact in self.service.result(run_id)["artifacts"] + } + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], + ["revise"], + ) + self.assertEqual(receipt["attempts"][0]["status"], "succeeded") + self.assertEqual(receipt["attempts"][1]["status"], "cancelled") + def test_headless_lead_is_a_second_fenced_attempt_and_terminalizes_automatically(self): self.configure_fixture_headless() started = self.service.start( diff --git a/test/core/test_store.py b/test/core/test_store.py index 354807a..98dd45c 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -232,7 +232,13 @@ def test_artifact_is_finalized_and_verified_before_reference(self): self.assertEqual(row[1], 13) @staticmethod - def routed_snapshot(wall_seconds=300): + def routed_snapshot( + wall_seconds=300, + *, + max_concurrency=1, + status="available", + unknown_capacity_policy="allow_bounded", + ): return { "task": {"budget": {"wall_seconds": wall_seconds}}, "routing": { @@ -244,7 +250,11 @@ def routed_snapshot(wall_seconds=300): }, }, "capacity": { - "shared-pool": {"max_concurrency": 1}, + "shared-pool": { + "max_concurrency": max_concurrency, + "status": status, + "unknown_capacity_policy": unknown_capacity_policy, + }, }, }, } @@ -309,6 +319,54 @@ def test_shared_pool_reservation_is_transactional_and_releases_on_finish(self): ) self.assertEqual(released.run_id, second.run_id) + def test_unknown_pool_allows_only_one_transactional_trial(self): + linked = self.root / "unknown-pool-linked" + subprocess.run( + ["git", "-C", str(self.repo), "worktree", "add", "--detach", "-q", + str(linked), "HEAD"], + check=True, + ) + snapshot = self.routed_snapshot(max_concurrency=3, status="unknown") + claims = [ + self.store.claim_start( + repository, + key, + {"task": {"budget": {"wall_seconds": 300}}}, + "owner", + ) + for repository, key in ( + (self.repo, "unknown-pool-first"), + (linked, "unknown-pool-second"), + ) + ] + versions = [ + self.store.complete_preparation( + claim.run_id, + claim.fencing_token, + snapshot, + worktree_path=str(repository), + ) + for claim, repository in zip(claims, (self.repo, linked)) + ] + self.store.reserve_attempt( + claims[0].run_id, + versions[0], + "unknown-supervisor-one", + "package", + "reviewer", + account_pool_id="shared-pool", + ) + with self.assertRaisesRegex(ConflictError, "account pool concurrency"): + self.store.reserve_attempt( + claims[1].run_id, + versions[1], + "unknown-supervisor-two", + "package", + "reviewer", + account_pool_id="shared-pool", + ) + self.assertEqual(self.store.active_pool_counts(), {"shared-pool": 1}) + def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): claim = self.store.claim_start( self.repo, From c7dbe9fe4a5eb1f65e51a0b8877f6281fddecd12 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 22 Sep 2026 20:46:48 +0530 Subject: [PATCH 066/197] docs: accept M3 branch review milestone --- docs/plans/engineering-team/M3-STATUS.md | 53 +++++++++++++++ docs/plans/engineering-team/RESUME.md | 65 +++++++++---------- docs/plans/engineering-team/backlog.json | 13 +++- .../evidence/M3-branch-review-2026-09-22.json | 58 +++++++++++++++++ 4 files changed, 153 insertions(+), 36 deletions(-) create mode 100644 docs/plans/engineering-team/M3-STATUS.md create mode 100644 docs/plans/engineering-team/evidence/M3-branch-review-2026-09-22.json diff --git a/docs/plans/engineering-team/M3-STATUS.md b/docs/plans/engineering-team/M3-STATUS.md new file mode 100644 index 0000000..12227d4 --- /dev/null +++ b/docs/plans/engineering-team/M3-STATUS.md @@ -0,0 +1,53 @@ +# M3 implementation status + +M3 is **complete** at implementation checkpoint `1737667`. The final gate +passes 188 core tests with `ResourceWarning` promoted to an error and all 10 +Bash regression files/202 assertions. The bounded live Codex review evidence +remains the successful subscription-backed run recorded at `9478796`; no +additional provider turn was used for the closeout. + +| Requirement | Evidence | Status | +|---|---|---| +| Deterministic selection | Versioned profile aliases, exact pins, explicit `none`/`policy` fallback, permission/billing filters and typed capacity produce stable frozen routing snapshots | verified offline | +| Frozen branch input | Base/target refs resolve to exact OIDs; committed config hashes, candidate hash and detached review/check workspaces remain stable while the submitted checkout, index and HEAD are preserved | verified offline | +| Reviewer evidence | Strict ordinary/adversarial prompts, review/check/evaluation schemas, candidate binding, read-only identity checks and malformed/denied/disconnected output faults prevent unsupported success | verified offline | +| Check and lead gates | Report-only failure stays visible without blocking acceptance; required failure blocks acceptance; host and headless leads support accept/reject/revise with bounded revisions | verified offline | +| Durable host handoff | Waiting JSON/Markdown packets, claim leases, renewal/takeover, stale completion fencing, replay and crash-resume continuation share the saved run ledger | verified offline | +| Terminal reporting | Success, rejection, preflight failure, worker failure, cancellation, waiting cancellation, timeout and budget exhaustion publish receipt JSON/Markdown, events, manifest and result receipt | verified offline | +| Headless leadership | Offline and native headless leads run as separate fenced attempts, verify their own frozen identity/evidence and terminalize without host intervention | verified offline | +| Runtime fallback | Reviewer and lead failures advance only through the frozen qualified fallback order; `fallback:none`, invocation/wall budgets and recovery-before-launch are enforced without profile substitution | verified offline | +| Capacity and accounting | Reservations are transactional, unknown pools permit one unresolved trial, live shared-pool reservations are observed, adapter launches are distinct from native usage and unavailable counts remain null | verified offline | +| Live branch review | Bundled Codex 0.153.4, gpt-5.5/low, ephemeral read-only execution found one supported regression, passed the required check and produced all five terminal report hashes | verified live | + +## Acceptance mapping + +- The live public branch review is bound to recorded base, target and candidate + hashes and produced an actionable supported finding. +- Offline gates separately prove a supported clean verdict, report-only and + required-check behavior, moving-ref stability, denied/missing output faults, + two-host fencing and source-checkout preservation. +- Receipts retain the effective profile/model/effort/toolbox snapshot, every + executed fallback profile, earlier review revisions and lead dispositions. +- `max_worker_invocations` counts durable runner launches, not reservations + abandoned before their launch fence. Revision, fallback and wall budgets all + end in explicit terminal reports rather than stranded resumable states. +- Standard and adversarial review prompts are distinct while sharing the same + read-only evidence and candidate-binding rules. + +## Independent audit closure + +The bounded M3 audit at `84deb77` found four defects: a recovered pre-launch +reservation consumed a fallback slot, headless-lead budget exhaustion could +strand a run, unknown capacity allowed more than one unresolved trial, and a +later failed/cancelled revision omitted earlier attempts and dispositions. +`1737667` fixes all four and adds direct reproductions. The complete 188-test +gate and 202 Bash assertions pass after those fixes. The requested follow-up +agent re-run could not start because its shared Plus window was exhausted; the +finding-specific regressions and full local gates are the closure evidence. + +## Boundary + +M3 proves a useful saved branch review from terminal/service APIs. It does not +claim M4 local-app MCP access, M5 issue implementation, M6 lifecycle learning, +M7 installation/all-surface receipts or C1 Council. Those remain required by +the full assignment. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 3f97cbe..686ea09 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,7 +2,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 17, 2026 +## Current position — September 22, 2026 - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` remains the published runtime @@ -22,32 +22,26 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne bounded target. `ddb6f51` fixes both: lease authorization now samples time after acquiring the SQLite write transaction, and public cancel resumes an interrupted `recovery_cleanup`. Both have deterministic regressions. -- M3 is in progress through `9478796`. `ca55990` adds strict deterministic - profile/policy routing, alias binding snapshots, pins/fallbacks, independent - reviewer selection and typed pool capacity. `1c7b614` freezes those exact - bytes and decisions during public preflight. `cd9a881` resolves base/target - OIDs and creates a detached run-owned review worktree without changing the - submitted checkout, index or HEAD. `a756307` defines strict review/check/ - evaluation evidence, `97d2c6c` runs the durable offline reviewer and trusted - checks, and `30cf49e` completes fenced host accept/revise/reject continuation, - bounded review retries and terminal JSON/Markdown/event/manifest reports. - `459ff3f` adds the public native Codex reviewer: it freezes the verified CLI, - launches one ephemeral read-only app-server turn inside the existing M2 - process group, verifies observed model/effort/sandbox identity, accepts only - strict candidate-bound JSON, and records native usage. `e10db94` pins the - compatible bundled Codex 0.153.4 runtime and isolates each review from old - sessions, config, plugins, skills and MCPs while exposing only existing - subscription auth. `9478796` makes the structured-output schema provider - compatible. The current gate is 169 core tests with `ResourceWarning` - promoted to failure plus 202 Bash assertions. +- M3 is accepted at `1737667`. The branch-review path now includes frozen + routing/input, native and offline reviewers, trusted checks, fenced host and + headless lead disposition, complete waiting/terminal reports, cumulative + budgets, transactional pool capacity and bounded frozen fallbacks. The final + gate is 188 core tests with `ResourceWarning` promoted to failure plus 202 + Bash assertions. See [M3-STATUS.md](M3-STATUS.md) and the + [portable closeout evidence](evidence/M3-branch-review-2026-09-22.json). - The bounded real public `branch-review` gate passed at `9478796` with verified gpt-5.5/low, read-only ephemeral execution, one supported finding, a passing required check, native-reported usage and all five terminal report hashes. See the [portable redacted evidence](evidence/M3-native-codex-review-2026-09-17.json). - M3 still needs rich reports for early terminal failures, materialized waiting - handoff reports and the configured headless-lead path before milestone - acceptance. -- Current provider readiness is external to M2: Claude CLI is not logged in; + That live receipt remains the M3 provider gate; closeout used offline tests + and did not consume another provider turn. +- The independent M3 audit at `84deb77` found four runtime/reporting defects. + `1737667` fixes all four with direct regressions: pre-launch recovery no + longer consumes a fallback/budget slot, headless lead exhaustion terminalizes, + unknown capacity allows one unresolved trial, and later failure/cancellation + retains earlier attempts and dispositions. A requested follow-up agent rerun + hit the shared Plus limit; the 188-test complete gate is green after the fixes. +- Last-observed provider readiness outside accepted M3: Claude CLI is not logged in; Grok CLI authentication expired; Gemini CLI's individual-account path is unsupported and its supported successor is Antigravity; Antigravity is authenticated but headless execution still lacks scoped permission/trust. @@ -60,7 +54,8 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne planning agents all hit the same Plus limit; continue locally until shared agent capacity is restored, then use only bounded leaf reviews. - Full assignment remains **M1–M7 plus C1**, as specified in - [SOL-HANDOFF.md](SOL-HANDOFF.md). M3 is next. + [SOL-HANDOFF.md](SOL-HANDOFF.md). M4 is next; M5 may proceed after the frozen + M3 service boundary and can be developed alongside M4 in isolated slices. ## Completed and preserved @@ -72,9 +67,9 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 169 tests passed through the provider-compatible M3 native Codex reviewer path | +| Python core discovery | 188 tests passed through M3 closeout with warnings promoted to errors | | Bash 3.2 regression suite | 10 test files, 202 assertions passed | -| Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 5 | +| Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 8 | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | | Integrated native adapter proof | Passed at `97a10f0`; gpt-5.5/low, read-only, correlated completion and confirmed process-group cleanup | | M2 crash/race matrix | Real subprocess interruptions plus independent-process start, writer, cancel, import and host-handoff races passed at `ddb6f51` | @@ -84,6 +79,7 @@ Verified at the implementation/evidence checkpoints above: | M3 host disposition/reporting | Accept/reject/revise, required-check blocking, retry budgets, stale claims, crash resume and five terminal reports passed at `30cf49e` | | M3 native Codex reviewer | Public start, exact identity verification, ephemeral read-only structured output, native usage and four provider-fault classes passed offline at `459ff3f` | | M3 live public review | Passed at `9478796`; gpt-5.5/low found one supported regression, the required check passed, host acceptance terminalized succeeded and five report hashes were retained | +| M3 closeout | Waiting/failure reports, headless leadership, cumulative budgets, live pool fencing, runtime fallbacks and all four independent-audit fixes pass at `1737667` | The first two saved-probe invocations failed before `Popen` because of probe-only path/field defects, so neither launched Codex nor consumed a model @@ -94,19 +90,20 @@ The successful run retained separate stderr files of 138,030 and 285,644 bytes, supporting the diagnosis that an undrained stderr pipe caused the earlier apparent nonresponses. -The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md) and -[M2-STATUS.md](M2-STATUS.md). [backlog.json](backlog.json) marks both complete -and M3 next. Unauthenticated, unsupported or permission-blocked provider paths -are not advertised as verified. +The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md), +[M2-STATUS.md](M2-STATUS.md) and [M3-STATUS.md](M3-STATUS.md). +[backlog.json](backlog.json) marks all three complete and M4 next. +Unauthenticated, unsupported or permission-blocked provider paths are not +advertised as verified. ## Exact next work 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Close the remaining M3 contract gaps: rich terminal reports for failures - before host disposition, materialized waiting handoff reports, and the - configured headless-lead path. Rerun the full core/Bash gates and audit M3 - against its acceptance section before marking it complete. +2. Execute M4 from [IMPLEMENTATION.md](IMPLEMENTATION.md): pin/test the optional + MCP SDK without coupling it to core CLI imports, map the saved-run service + operations to strict stdio tools, enforce worker recursion guards, and add + idempotent local integration/doctor evidence. 3. Keep Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. They do not block the independent Codex M3 gate. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 7717f64..b60ed4d 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -11,7 +11,7 @@ "execution_brief": "SOL-HANDOFF.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", - "next_milestone": "M3", + "next_milestone": "M4", "milestones": [ { "id": "M1", @@ -82,7 +82,7 @@ "id": "M3", "title": "Usable branch review from terminal", "depends_on": ["M2"], - "status": "in_progress", + "status": "complete", "acceptance_section": "M3 — Ship a useful branch review", "evidence": [ { @@ -156,6 +156,15 @@ "artifact": "evidence/M3-native-codex-review-2026-09-17.json", "recorded_at": "2026-09-17T21:34:40+05:30", "availability": "portable_redacted" + }, + { + "kind": "milestone_acceptance", + "revision": "1737667", + "command_or_action": "188 core tests with ResourceWarning promoted to error, 202 shell assertions, bounded reviewer/lead fallback, headless budget terminalization, transactional unknown-capacity trial limit and independent audit closure", + "outcome": "All M3 branch-review, routing, selection, accounting, headless-lead, terminal-report and live Codex acceptance gates are covered; four independent-audit findings are fixed with direct regressions", + "artifact": "evidence/M3-branch-review-2026-09-22.json", + "recorded_at": "2026-09-22T20:42:05+05:30", + "availability": "portable_redacted" } ], "blocker": null diff --git a/docs/plans/engineering-team/evidence/M3-branch-review-2026-09-22.json b/docs/plans/engineering-team/evidence/M3-branch-review-2026-09-22.json new file mode 100644 index 0000000..02f29c8 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M3-branch-review-2026-09-22.json @@ -0,0 +1,58 @@ +{ + "schema_version": 1, + "milestone": "M3", + "status": "complete", + "implementation_revision": "1737667a3a847bfd09559abd5a53f8ec88745982", + "recorded_at": "2026-09-22T20:42:05+05:30", + "offline_gate": { + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONPATH=plugin/core/src python3 -W error::ResourceWarning -m unittest discover -s test/core -q", + "result": "188 tests passed", + "bash_command": "bash test/run.sh", + "bash_result": "10 test files and 202 assertions passed" + }, + "live_gate": { + "revision": "9478796", + "artifact": "M3-native-codex-review-2026-09-17.json", + "result": "subscription-backed Codex gpt-5.5/low read-only review succeeded with one supported finding, a passing required check, 53700 native-reported tokens and five terminal report hashes" + }, + "closeout_checkpoints": [ + { + "revision": "9dc795a", + "result": "waiting handoff and rich early-terminal report sets" + }, + { + "revision": "8f3fa56", + "result": "durable host and headless lead paths with distinct fenced attempts" + }, + { + "revision": "e1afa4d", + "result": "cumulative wall budgets, transactional live pool reservations and complete waiting cancellation reports" + }, + { + "revision": "84deb77", + "result": "bounded frozen reviewer and lead runtime fallbacks with schema-8 attempt identity" + }, + { + "revision": "1737667", + "result": "four independent-audit defects fixed with deterministic regressions" + } + ], + "independent_audit": { + "target_revision": "84deb77", + "findings": [ + "pre-launch recovery consumed an unused fallback slot", + "headless-lead invocation exhaustion stranded an awaiting run", + "unknown capacity allowed more than one unresolved trial", + "later early-terminal reports omitted completed earlier revisions" + ], + "closure_revision": "1737667", + "closure": "all four reproduced by tracked tests; focused and complete offline gates pass", + "follow_up_limit": "the same audit agent could not rerun after the fixes because the shared Plus window was exhausted" + }, + "provider_blockers_outside_m3": [ + "Claude CLI is not logged in", + "Grok authentication is expired", + "Antigravity is authenticated but headless scoped trust/permission remains blocked" + ], + "residual_m3_blockers": [] +} From 243cb245d2044d9810bb0df22fc113a627c4771d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 22 Sep 2026 20:57:20 +0530 Subject: [PATCH 067/197] feat: establish optional MCP transport boundary --- plugin/core/pyproject.toml | 3 + plugin/core/requirements-mcp.lock | 32 ++++++ plugin/core/src/devsquad/cli.py | 21 ++++ plugin/core/src/devsquad/mcp_server.py | 48 +++++++++ test/core/test_mcp.py | 130 +++++++++++++++++++++++++ 5 files changed, 234 insertions(+) create mode 100644 plugin/core/requirements-mcp.lock create mode 100644 plugin/core/src/devsquad/mcp_server.py create mode 100644 test/core/test_mcp.py diff --git a/plugin/core/pyproject.toml b/plugin/core/pyproject.toml index f9bf72e..b28e8cb 100644 --- a/plugin/core/pyproject.toml +++ b/plugin/core/pyproject.toml @@ -8,6 +8,9 @@ version = "0.1.0" requires-python = ">=3.11" dependencies = [] +[project.optional-dependencies] +mcp = ["mcp==2.2.0"] + [project.scripts] squad = "devsquad.cli:main" diff --git a/plugin/core/requirements-mcp.lock b/plugin/core/requirements-mcp.lock new file mode 100644 index 0000000..7040407 --- /dev/null +++ b/plugin/core/requirements-mcp.lock @@ -0,0 +1,32 @@ +# DevSquad optional local MCP transport. +# Resolved from mcp==2.2.0 on 2026-09-22; install this lock only in the +# isolated MCP environment. The base devsquad-core package has no runtime +# dependencies and must remain independently installable. +annotated-types==0.8.0 +anyio==4.15.1 +attrs==26.1.0 +cffi==2.1.1 +click==8.5.0 +cryptography==50.0.1 +h11==0.16.0 +httpcore2==2.13.0 +httpx2==2.13.0 +idna==3.20 +jsonschema==4.26.0 +jsonschema-specifications==2025.9.1 +mcp==2.2.0 +mcp-types==2.2.0 +opentelemetry-api==1.44.0 +pycparser==3.0 +pydantic==2.13.5 +pydantic_core==2.46.5 +PyJWT==2.14.0 +python-multipart==0.0.32 +referencing==0.37.0 +rpds-py==2026.6.3 +sse-starlette==3.4.11 +starlette==1.6.0 +truststore==0.10.4 +typing-inspection==0.4.4 +typing_extensions==4.16.0 +uvicorn==0.53.0 diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 018d7c5..a482966 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -138,6 +138,20 @@ def command_handoff_complete(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).handoff_complete(args.run, claim, decision)), 0 +def command_mcp_serve(args: argparse.Namespace) -> int: + # Keep this import inside the explicitly requested command. Importing the + # ordinary CLI must remain valid when the optional SDK is absent. + from .mcp_server import MCPDependencyUnavailable, serve_stdio + + try: + serve_stdio(Path(args.runtime_dir)) + except MCPDependencyUnavailable as exc: + # stdout is the MCP protocol channel, including during startup. + print(str(exc), file=sys.stderr) + return 69 + return 0 + + class ContractParser(argparse.ArgumentParser): def error(self, message: str) -> None: raise ContractError(message) @@ -189,12 +203,19 @@ def parser() -> argparse.ArgumentParser: complete.add_argument("--json", action="store_true") complete.add_argument("--runtime-dir", default=runtime_default) complete.set_defaults(func=command_handoff_complete) + mcp = sub.add_parser("mcp") + mcp_sub = mcp.add_subparsers(dest="mcp_command", required=True) + serve = mcp_sub.add_parser("serve") + serve.add_argument("--runtime-dir", default=runtime_default) + serve.set_defaults(stream_func=command_mcp_serve) return p def main(argv: list[str] | None = None) -> int: try: args = parser().parse_args(argv) + if hasattr(args, "stream_func"): + return args.stream_func(args) response, code = args.func(args) print(json.dumps(response, sort_keys=True)) return code diff --git a/plugin/core/src/devsquad/mcp_server.py b/plugin/core/src/devsquad/mcp_server.py new file mode 100644 index 0000000..e1920e5 --- /dev/null +++ b/plugin/core/src/devsquad/mcp_server.py @@ -0,0 +1,48 @@ +"""Optional local MCP transport for the durable DevSquad service. + +This module must remain importable without the MCP SDK. Ordinary CLI and +worker processes never pay for or depend on the optional transport package. +""" + +from __future__ import annotations + +from pathlib import Path +from typing import Any + +MCP_SDK_REQUIREMENT = "mcp==2.2.0" + + +class MCPDependencyUnavailable(RuntimeError): + """Raised when the explicitly requested MCP transport is not installed.""" + + +def _server_type() -> Any: + try: + from mcp.server import MCPServer + except ModuleNotFoundError as exc: + if exc.name != "mcp": + raise + raise MCPDependencyUnavailable( + "DevSquad MCP support is not installed; install the optional " + f"dependency with: python3 -m pip install 'devsquad-core[mcp]' " + f"(requires {MCP_SDK_REQUIREMENT})" + ) from exc + return MCPServer + + +def build_server(runtime: Path) -> Any: + """Build the stdio server without starting it. + + Tools are registered by later M4 slices. Keeping construction separate + makes the optional dependency and installed-wheel boundary testable. + """ + + del runtime + server_type = _server_type() + return server_type("DevSquad") + + +def serve_stdio(runtime: Path) -> None: + """Run the local MCP server; stdout is owned exclusively by the SDK.""" + + build_server(runtime).run() diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py new file mode 100644 index 0000000..491071e --- /dev/null +++ b/test/core/test_mcp.py @@ -0,0 +1,130 @@ +import contextlib +import io +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +import tomllib +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +CORE = ROOT / "plugin/core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad import cli, mcp_server + + +class MCPDependencyBoundaryTest(unittest.TestCase): + def test_supported_sdk_is_an_exact_optional_dependency(self): + project = tomllib.loads((CORE / "pyproject.toml").read_text())["project"] + self.assertEqual(project["requires-python"], ">=3.11") + self.assertEqual(project["dependencies"], []) + self.assertEqual(project["optional-dependencies"]["mcp"], ["mcp==2.2.0"]) + self.assertEqual(mcp_server.MCP_SDK_REQUIREMENT, "mcp==2.2.0") + locked = { + line.strip().lower() + for line in (CORE / "requirements-mcp.lock").read_text().splitlines() + if line.strip() and not line.startswith("#") + } + self.assertIn("mcp==2.2.0", locked) + + def test_core_cli_import_does_not_import_optional_sdk(self): + probe = """ +import sys +sys.path.insert(0, {source!r}) +import devsquad.cli +assert not any(name == 'mcp' or name.startswith('mcp.') for name in sys.modules) +""".format(source=str(CORE / "src")) + subprocess.run([sys.executable, "-P", "-c", probe], check=True) + + def test_mcp_serve_dispatches_without_writing_protocol_stdout(self): + runtime = Path("/tmp/devsquad-mcp-boundary") + stdout, stderr = io.StringIO(), io.StringIO() + with mock.patch.object(mcp_server, "serve_stdio") as serve: + with contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + code = cli.main(["mcp", "serve", "--runtime-dir", str(runtime)]) + self.assertEqual((code, stdout.getvalue(), stderr.getvalue()), (0, "", "")) + serve.assert_called_once_with(runtime) + + def test_missing_sdk_is_actionable_and_keeps_stdout_clean(self): + stdout, stderr = io.StringIO(), io.StringIO() + missing = mcp_server.MCPDependencyUnavailable("install devsquad-core[mcp]") + with mock.patch.object(mcp_server, "serve_stdio", side_effect=missing): + with contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + code = cli.main(["mcp", "serve"]) + self.assertEqual(code, 69) + self.assertEqual(stdout.getvalue(), "") + self.assertIn("devsquad-core[mcp]", stderr.getvalue()) + + +class InstalledWheelMCPBoundaryTest(unittest.TestCase): + @staticmethod + def build_python(): + candidates = [ + os.environ.get("DEVSQUAD_BUILD_PYTHON"), + sys.executable, + str(Path.home() / ".cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3"), + shutil.which("python3.13"), + shutil.which("python3.12"), + shutil.which("python3.11"), + ] + for candidate in dict.fromkeys(value for value in candidates if value): + result = subprocess.run( + [candidate, "-c", "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68"], + text=True, + capture_output=True, + ) + if result.returncode == 0: + return candidate + return None + + def test_plain_installed_wheel_keeps_cli_usable_without_mcp(self): + build_python = self.build_python() + if build_python is None: + self.skipTest("offline wheel gate requires setuptools>=68 and wheel") + with tempfile.TemporaryDirectory(prefix="devsquad-mcp-wheel-") as directory: + root = Path(directory) + source = root / "core" + shutil.copytree(CORE, source) + wheels = root / "wheels" + wheels.mkdir() + subprocess.run( + [ + build_python, "-m", "pip", "wheel", str(source), + "--wheel-dir", str(wheels), "--no-index", "--no-deps", "--no-build-isolation", + ], + check=True, + text=True, + capture_output=True, + ) + wheel = next(wheels.glob("devsquad_core-*.whl")) + environment = os.environ.copy() + environment.pop("PYTHONPATH", None) + venv = root / "venv" + subprocess.run([build_python, "-m", "venv", str(venv)], check=True, env=environment) + python = venv / ("Scripts/python.exe" if os.name == "nt" else "bin/python") + squad = venv / ("Scripts/squad.exe" if os.name == "nt" else "bin/squad") + subprocess.run( + [str(python), "-m", "pip", "install", "--no-index", "--no-deps", str(wheel)], + check=True, + text=True, + capture_output=True, + env=environment, + ) + version = subprocess.run( + [str(squad), "--version"], check=True, text=True, capture_output=True, env=environment, + ) + self.assertEqual(version.stdout.strip(), "squad 0.1.0") + missing = subprocess.run( + [str(squad), "mcp", "serve"], text=True, capture_output=True, env=environment, + ) + self.assertEqual(missing.returncode, 69) + self.assertEqual(missing.stdout, "") + self.assertIn("devsquad-core[mcp]", missing.stderr) + + +if __name__ == "__main__": + unittest.main() From b2e842e360d669132c88094a894047c5ecc9cd43 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 22 Sep 2026 21:06:51 +0530 Subject: [PATCH 068/197] feat: expose durable runs over MCP --- plugin/core/src/devsquad/mcp_server.py | 218 ++++++++++++++++++++++++- test/core/test_mcp.py | 178 ++++++++++++++++++++ 2 files changed, 388 insertions(+), 8 deletions(-) diff --git a/plugin/core/src/devsquad/mcp_server.py b/plugin/core/src/devsquad/mcp_server.py index e1920e5..bfd1243 100644 --- a/plugin/core/src/devsquad/mcp_server.py +++ b/plugin/core/src/devsquad/mcp_server.py @@ -7,9 +7,15 @@ from __future__ import annotations from pathlib import Path -from typing import Any +from typing import Any, Callable + +from . import __version__ +from .contracts import ContractError, envelope, error_payload +from .service import Service +from .store import ConflictError MCP_SDK_REQUIREMENT = "mcp==2.2.0" +MAX_ARTIFACT_PREVIEW_BYTES = 16 * 1024 class MCPDependencyUnavailable(RuntimeError): @@ -30,16 +36,212 @@ def _server_type() -> Any: return MCPServer -def build_server(runtime: Path) -> Any: - """Build the stdio server without starting it. +class MCPBridge: + """Strict MCP-facing application functions over one saved runtime.""" + + def __init__(self, runtime: Path, service: Service | None = None): + self.runtime = runtime.resolve() + self.service = service or Service(runtime) + + @staticmethod + def _response(operation: Callable[[], dict[str, Any]]) -> dict[str, Any]: + try: + return envelope(data=operation()) + except ContractError as exc: + return envelope(error=error_payload(exc.code, str(exc))) + except Exception as exc: + return envelope(error=error_payload("INTERNAL_ERROR", str(exc))) + + def start( + self, + task: dict[str, Any], + idempotency_key: str, + supersedes_run_id: str | None = None, + ) -> dict[str, Any]: + """Validate and save a run, returning without observing its worker.""" + + return self._response( + lambda: self.service.start(task, idempotency_key, supersedes_run_id) + ) + + def status(self, run_id: str) -> dict[str, Any]: + """Inspect the current projection for a saved run.""" + + return self._response(lambda: self.service.status(run_id)) + + def events( + self, run_id: str, after: int = 0, limit: int = 100, + ) -> dict[str, Any]: + """Read one bounded event page using its durable integer cursor.""" + + return self._response(lambda: self.service.events(run_id, after, limit)) + + def result( + self, run_id: str, preview_bytes: int = 4096, + ) -> dict[str, Any]: + """Return receipt references and bounded UTF-8 artifact previews.""" + + def operation() -> dict[str, Any]: + if (type(preview_bytes) is not int + or not 0 <= preview_bytes <= MAX_ARTIFACT_PREVIEW_BYTES): + raise ContractError( + "preview_bytes must be between 0 and " + f"{MAX_ARTIFACT_PREVIEW_BYTES}" + ) + result = self.service.result(run_id) + if not result.get("ready") or preview_bytes == 0: + return result + remaining = preview_bytes + artifact_root = (self.runtime / "artifacts").resolve(strict=True) + artifacts = [] + for saved in result.get("artifacts", []): + artifact = dict(saved) + artifact["preview_text"] = None + artifact["preview_truncated"] = artifact.get("byte_size", 0) > 0 + if remaining > 0: + path = Path(artifact["path"]).resolve(strict=True) + try: + path.relative_to(artifact_root) + except ValueError as exc: + raise ConflictError( + "result artifact path escapes the saved runtime" + ) from exc + with path.open("rb") as stream: + raw = stream.read(remaining + 1) + consumed = min(len(raw), remaining) + try: + artifact["preview_text"] = raw[:consumed].decode("utf-8") + except UnicodeDecodeError: + artifact["preview_text"] = None + artifact["preview_truncated"] = len(raw) > consumed + remaining -= consumed + artifacts.append(artifact) + return {**result, "artifacts": artifacts, "preview_bytes": preview_bytes} + + return self._response(operation) + + def cancel(self, run_id: str) -> dict[str, Any]: + """Persist cancellation intent without observing worker completion.""" + + return self._response(lambda: self.service.cancel(run_id)) + + def resume( + self, run_id: str, recovery: dict[str, Any] | None = None, + ) -> dict[str, Any]: + """Reconcile and safely resume a saved run.""" + + return self._response(lambda: self.service.resume(run_id, recovery)) + + def handoff_claim( + self, + run_id: str, + expected_version: int, + owner: str, + prior_claim: dict[str, Any] | None = None, + ) -> dict[str, Any]: + """Claim or renew one fenced host handoff.""" + + return self._response( + lambda: self.service.handoff_claim( + run_id, expected_version, owner, prior_claim, + ) + ) + + def handoff_complete( + self, + run_id: str, + claim: dict[str, Any], + decision: dict[str, Any], + ) -> dict[str, Any]: + """Submit a decision against a current fenced host claim.""" - Tools are registered by later M4 slices. Keeping construction separate - makes the optional dependency and installed-wheel boundary testable. - """ + return self._response( + lambda: self.service.handoff_complete(run_id, claim, decision) + ) + + +def build_server(runtime: Path, service: Service | None = None) -> Any: + """Build the local stdio server without starting it.""" - del runtime server_type = _server_type() - return server_type("DevSquad") + bridge = MCPBridge(runtime, service) + server = server_type( + "DevSquad", + version=__version__, + instructions=( + "Operate on durable local DevSquad runs. Submit a task once with " + "an idempotency key, then inspect status/events/result by run ID." + ), + ) + + @server.tool(name="squad_start") + def squad_start( + task: dict[str, Any], + idempotency_key: str, + supersedes_run_id: str | None = None, + ) -> dict[str, Any]: + """Validate, snapshot and persist one durable DevSquad run.""" + + return bridge.start(task, idempotency_key, supersedes_run_id) + + @server.tool(name="squad_status") + def squad_status(run_id: str) -> dict[str, Any]: + """Inspect state, version, active work and the next action.""" + + return bridge.status(run_id) + + @server.tool(name="squad_events") + def squad_events( + run_id: str, after: int = 0, limit: int = 100, + ) -> dict[str, Any]: + """Read a bounded durable event page after an integer cursor.""" + + return bridge.events(run_id, after, limit) + + @server.tool(name="squad_result") + def squad_result( + run_id: str, preview_bytes: int = 4096, + ) -> dict[str, Any]: + """Read result references and capped local artifact previews.""" + + return bridge.result(run_id, preview_bytes) + + @server.tool(name="squad_cancel") + def squad_cancel(run_id: str) -> dict[str, Any]: + """Persist cancellation intent for a saved run.""" + + return bridge.cancel(run_id) + + @server.tool(name="squad_resume") + def squad_resume( + run_id: str, recovery: dict[str, Any] | None = None, + ) -> dict[str, Any]: + """Reconcile ownership and safely resume a saved run.""" + + return bridge.resume(run_id, recovery) + + @server.tool(name="squad_handoff_claim") + def squad_handoff_claim( + run_id: str, + expected_version: int, + owner: str, + prior_claim: dict[str, Any] | None = None, + ) -> dict[str, Any]: + """Obtain or renew a fenced host claim and its input packet.""" + + return bridge.handoff_claim(run_id, expected_version, owner, prior_claim) + + @server.tool(name="squad_handoff_complete") + def squad_handoff_complete( + run_id: str, + claim: dict[str, Any], + decision: dict[str, Any], + ) -> dict[str, Any]: + """Submit a host disposition against a current fenced claim.""" + + return bridge.handoff_complete(run_id, claim, decision) + + return server def serve_stdio(runtime: Path) -> None: diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index 491071e..04b94c3 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -1,5 +1,7 @@ import contextlib +import asyncio import io +import importlib.util import os from pathlib import Path import shutil @@ -15,6 +17,8 @@ sys.path.insert(0, str(CORE / "src")) from devsquad import cli, mcp_server +from devsquad.contracts import ContractError +from devsquad.store import ConflictError class MCPDependencyBoundaryTest(unittest.TestCase): @@ -60,6 +64,180 @@ def test_missing_sdk_is_actionable_and_keeps_stdout_clean(self): self.assertIn("devsquad-core[mcp]", stderr.getvalue()) +class MCPBridgeTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-mcp-bridge-") + self.runtime = Path(self.temp.name) / "runtime" + (self.runtime / "artifacts").mkdir(parents=True) + self.service = mock.Mock() + self.bridge = mcp_server.MCPBridge(self.runtime, self.service) + + def tearDown(self): + self.temp.cleanup() + + def assert_success(self, payload, data): + self.assertEqual(payload, { + "schema_version": 1, + "ok": True, + "data": data, + "error": None, + }) + + def test_operations_map_directly_to_the_saved_run_service(self): + task = {"schema_version": 1} + recovery = {"attempt_id": "attempt-1", "disposition": "confirm_dead"} + claim = {"run_id": "run-1", "fencing_token": 3} + decision = {"disposition": "accept"} + calls = [ + ( + lambda: self.bridge.start(task, "key-1", "run-0"), + "start", (task, "key-1", "run-0"), + {"run_id": "run-1", "state": "queued"}, + ), + ( + lambda: self.bridge.status("run-1"), + "status", ("run-1",), {"run_id": "run-1", "state": "running"}, + ), + ( + lambda: self.bridge.events("run-1", 4, 25), + "events", ("run-1", 4, 25), {"events": [], "next_cursor": 4}, + ), + ( + lambda: self.bridge.cancel("run-1"), + "cancel", ("run-1",), {"run_id": "run-1", "state": "cancelling"}, + ), + ( + lambda: self.bridge.resume("run-1", recovery), + "resume", ("run-1", recovery), {"run_id": "run-1", "state": "running"}, + ), + ( + lambda: self.bridge.handoff_claim("run-1", 7, "claude", claim), + "handoff_claim", ("run-1", 7, "claude", claim), + {"run_id": "run-1", "action": "renewed"}, + ), + ( + lambda: self.bridge.handoff_complete("run-1", claim, decision), + "handoff_complete", ("run-1", claim, decision), + {"run_id": "run-1", "state": "succeeded"}, + ), + ] + for invoke, method_name, expected_args, response in calls: + with self.subTest(operation=method_name): + method = getattr(self.service, method_name) + method.return_value = response + self.assert_success(invoke(), response) + method.assert_called_once_with(*expected_args) + method.reset_mock() + + def test_contract_conflict_and_internal_failures_keep_machine_envelopes(self): + cases = [ + (ContractError("bad request"), "INPUT_INVALID"), + (ConflictError("stale claim"), "CONFLICT"), + (OSError("disk unavailable"), "INTERNAL_ERROR"), + ] + for failure, code in cases: + with self.subTest(code=code): + self.service.status.side_effect = failure + payload = self.bridge.status("run-1") + self.assertEqual(payload["schema_version"], 1) + self.assertFalse(payload["ok"]) + self.assertIsNone(payload["data"]) + self.assertEqual(payload["error"]["code"], code) + self.assertEqual(payload["error"]["message"], str(failure)) + + def test_result_caps_preview_bytes_across_saved_artifacts(self): + first = self.runtime / "artifacts" / "receipt.json" + second = self.runtime / "artifacts" / "events.jsonl" + first.write_text("abcdef") + second.write_text("ghijkl") + artifacts = [ + {"id": "a-1", "name": first.name, "path": str(first), "sha256": "1" * 64, "byte_size": 6}, + {"id": "a-2", "name": second.name, "path": str(second), "sha256": "2" * 64, "byte_size": 6}, + ] + self.service.result.return_value = { + "run_id": "run-1", "ready": True, "state": "succeeded", "artifacts": artifacts, + } + payload = self.bridge.result("run-1", preview_bytes=5) + self.assertTrue(payload["ok"]) + result = payload["data"] + self.assertEqual(result["preview_bytes"], 5) + self.assertEqual(result["artifacts"][0]["preview_text"], "abcde") + self.assertTrue(result["artifacts"][0]["preview_truncated"]) + self.assertIsNone(result["artifacts"][1]["preview_text"]) + self.assertTrue(result["artifacts"][1]["preview_truncated"]) + self.assertEqual( + {key: result["artifacts"][0][key] for key in ("id", "path", "sha256")}, + {"id": "a-1", "path": str(first), "sha256": "1" * 64}, + ) + + too_large = self.bridge.result( + "run-1", preview_bytes=mcp_server.MAX_ARTIFACT_PREVIEW_BYTES + 1, + ) + self.assertFalse(too_large["ok"]) + self.assertEqual(too_large["error"]["code"], "INPUT_INVALID") + self.assertEqual(self.service.result.call_count, 1) + + def test_result_rejects_preview_path_outside_the_runtime(self): + outside = Path(self.temp.name) / "outside.txt" + outside.write_text("not a saved runtime artifact") + self.service.result.return_value = { + "run_id": "run-1", + "ready": True, + "state": "succeeded", + "artifacts": [{ + "id": "a-1", "name": outside.name, "path": str(outside), + "sha256": "1" * 64, "byte_size": outside.stat().st_size, + }], + } + payload = self.bridge.result("run-1") + self.assertFalse(payload["ok"]) + self.assertEqual(payload["error"]["code"], "CONFLICT") + self.assertIn("escapes", payload["error"]["message"]) + + +@unittest.skipUnless(importlib.util.find_spec("mcp"), "optional MCP SDK is not installed") +class OfficialSDKConformanceTest(unittest.TestCase): + def test_tool_schemas_and_calls_use_the_official_in_memory_transport(self): + from mcp import Client + + service = mock.Mock() + service.status.return_value = {"run_id": "run-1", "state": "running", "version": 2} + service.start.return_value = {"run_id": "run-1", "state": "queued", "created": True} + service.cancel.return_value = {"run_id": "run-1", "state": "cancelling", "version": 3} + with tempfile.TemporaryDirectory(prefix="devsquad-sdk-server-") as directory: + server = mcp_server.build_server(Path(directory), service) + + async def probe(): + async with Client(server) as client: + listing = await client.list_tools() + tools = {tool.name: tool for tool in listing.tools} + self.assertEqual(set(tools), { + "squad_start", "squad_status", "squad_events", "squad_result", + "squad_cancel", "squad_resume", "squad_handoff_claim", + "squad_handoff_complete", + }) + self.assertEqual( + tools["squad_status"].input_schema["required"], ["run_id"], + ) + status = await client.call_tool("squad_status", {"run_id": "run-1"}) + self.assertFalse(status.is_error) + self.assertEqual(status.structured_content["data"]["version"], 2) + started = await client.call_tool("squad_start", { + "task": {"schema_version": 1}, "idempotency_key": "key-1", + }) + self.assertFalse(started.is_error) + self.assertEqual(started.structured_content["data"]["run_id"], "run-1") + cancelled = await client.call_tool("squad_cancel", {"run_id": "run-1"}) + self.assertFalse(cancelled.is_error) + self.assertEqual(cancelled.structured_content["data"]["state"], "cancelling") + malformed = await client.call_tool("squad_events", { + "run_id": "run-1", "after": 0, "limit": "not-an-integer", + }) + self.assertTrue(malformed.is_error) + + asyncio.run(probe()) + + class InstalledWheelMCPBoundaryTest(unittest.TestCase): @staticmethod def build_python(): From 643910d7b8b821aebcb7242c3690bc0f5aa28b16 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 22 Sep 2026 21:13:58 +0530 Subject: [PATCH 069/197] fix: block recursive DevSquad workers --- plugin/core/src/devsquad/cli.py | 8 +- plugin/core/src/devsquad/mcp_server.py | 102 +++++++++++++++++++++---- plugin/hooks/scripts/pre-compact.sh | 6 +- plugin/hooks/scripts/pre-tool-use.sh | 6 +- plugin/hooks/scripts/session-start.sh | 6 +- plugin/hooks/scripts/stop.sh | 6 +- test/core/test_mcp.py | 82 +++++++++++++++++++- test/test_hooks.sh | 30 ++++++++ 8 files changed, 221 insertions(+), 25 deletions(-) diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index a482966..4f6cc2a 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -144,7 +144,11 @@ def command_mcp_serve(args: argparse.Namespace) -> int: from .mcp_server import MCPDependencyUnavailable, serve_stdio try: - serve_stdio(Path(args.runtime_dir)) + serve_stdio( + Path(args.runtime_dir), + caller_surface=args.surface, + caller_session_ref=args.session_ref, + ) except MCPDependencyUnavailable as exc: # stdout is the MCP protocol channel, including during startup. print(str(exc), file=sys.stderr) @@ -207,6 +211,8 @@ def parser() -> argparse.ArgumentParser: mcp_sub = mcp.add_subparsers(dest="mcp_command", required=True) serve = mcp_sub.add_parser("serve") serve.add_argument("--runtime-dir", default=runtime_default) + serve.add_argument("--surface") + serve.add_argument("--session-ref") serve.set_defaults(stream_func=command_mcp_serve) return p diff --git a/plugin/core/src/devsquad/mcp_server.py b/plugin/core/src/devsquad/mcp_server.py index bfd1243..9516f15 100644 --- a/plugin/core/src/devsquad/mcp_server.py +++ b/plugin/core/src/devsquad/mcp_server.py @@ -6,11 +6,13 @@ from __future__ import annotations +import copy +import os from pathlib import Path -from typing import Any, Callable +from typing import Any, Callable, Mapping from . import __version__ -from .contracts import ContractError, envelope, error_payload +from .contracts import ContractError, PolicyDenied, envelope, error_payload from .service import Service from .store import ConflictError @@ -39,9 +41,26 @@ def _server_type() -> Any: class MCPBridge: """Strict MCP-facing application functions over one saved runtime.""" - def __init__(self, runtime: Path, service: Service | None = None): + def __init__( + self, + runtime: Path, + service: Service | None = None, + *, + caller_surface: str | None = None, + caller_session_ref: str | None = None, + environment: Mapping[str, str] | None = None, + ): + for label, value in ( + ("caller_surface", caller_surface), + ("caller_session_ref", caller_session_ref), + ): + if value is not None and (not isinstance(value, str) or not value): + raise ContractError(f"{label} must be a non-empty string or null") self.runtime = runtime.resolve() self.service = service or Service(runtime) + self.caller_surface = caller_surface + self.caller_session_ref = caller_session_ref + self.environment = os.environ if environment is None else environment @staticmethod def _response(operation: Callable[[], dict[str, Any]]) -> dict[str, Any]: @@ -52,6 +71,37 @@ def _response(operation: Callable[[], dict[str, Any]]) -> dict[str, Any]: except Exception as exc: return envelope(error=error_payload("INTERNAL_ERROR", str(exc))) + def _mutation_response( + self, operation: Callable[[], dict[str, Any]], + ) -> dict[str, Any]: + def guarded() -> dict[str, Any]: + worker = self.environment.get("DEVSQUAD_WORKER", "0") + depth = self.environment.get("DEVSQUAD_DELEGATION_DEPTH", "0") + if worker not in {"", "0"} or depth not in {"", "0"}: + raise PolicyDenied( + "DevSquad worker sessions cannot start or mutate team workflows" + ) + return operation() + + return self._response(guarded) + + def _task_with_bound_origin(self, task: dict[str, Any]) -> dict[str, Any]: + if self.caller_surface is None and self.caller_session_ref is None: + return task + bound = copy.deepcopy(task) + if not isinstance(bound, dict): + raise ContractError("task must be an object") + saved_origin = bound.get("origin", {}) + if not isinstance(saved_origin, dict): + raise ContractError("task origin must be an object") + origin = dict(saved_origin) + if self.caller_surface is not None: + origin["surface"] = self.caller_surface + if self.caller_session_ref is not None: + origin["session_ref"] = self.caller_session_ref + bound["origin"] = origin + return bound + def start( self, task: dict[str, Any], @@ -60,8 +110,12 @@ def start( ) -> dict[str, Any]: """Validate and save a run, returning without observing its worker.""" - return self._response( - lambda: self.service.start(task, idempotency_key, supersedes_run_id) + return self._mutation_response( + lambda: self.service.start( + self._task_with_bound_origin(task), + idempotency_key, + supersedes_run_id, + ) ) def status(self, run_id: str) -> dict[str, Any]: @@ -123,14 +177,14 @@ def operation() -> dict[str, Any]: def cancel(self, run_id: str) -> dict[str, Any]: """Persist cancellation intent without observing worker completion.""" - return self._response(lambda: self.service.cancel(run_id)) + return self._mutation_response(lambda: self.service.cancel(run_id)) def resume( self, run_id: str, recovery: dict[str, Any] | None = None, ) -> dict[str, Any]: """Reconcile and safely resume a saved run.""" - return self._response(lambda: self.service.resume(run_id, recovery)) + return self._mutation_response(lambda: self.service.resume(run_id, recovery)) def handoff_claim( self, @@ -141,7 +195,7 @@ def handoff_claim( ) -> dict[str, Any]: """Claim or renew one fenced host handoff.""" - return self._response( + return self._mutation_response( lambda: self.service.handoff_claim( run_id, expected_version, owner, prior_claim, ) @@ -155,16 +209,29 @@ def handoff_complete( ) -> dict[str, Any]: """Submit a decision against a current fenced host claim.""" - return self._response( + return self._mutation_response( lambda: self.service.handoff_complete(run_id, claim, decision) ) -def build_server(runtime: Path, service: Service | None = None) -> Any: +def build_server( + runtime: Path, + service: Service | None = None, + *, + caller_surface: str | None = None, + caller_session_ref: str | None = None, + environment: Mapping[str, str] | None = None, +) -> Any: """Build the local stdio server without starting it.""" server_type = _server_type() - bridge = MCPBridge(runtime, service) + bridge = MCPBridge( + runtime, + service, + caller_surface=caller_surface, + caller_session_ref=caller_session_ref, + environment=environment, + ) server = server_type( "DevSquad", version=__version__, @@ -244,7 +311,16 @@ def squad_handoff_complete( return server -def serve_stdio(runtime: Path) -> None: +def serve_stdio( + runtime: Path, + *, + caller_surface: str | None = None, + caller_session_ref: str | None = None, +) -> None: """Run the local MCP server; stdout is owned exclusively by the SDK.""" - build_server(runtime).run() + build_server( + runtime, + caller_surface=caller_surface, + caller_session_ref=caller_session_ref, + ).run() diff --git a/plugin/hooks/scripts/pre-compact.sh b/plugin/hooks/scripts/pre-compact.sh index 372cd0a..e37d7ce 100755 --- a/plugin/hooks/scripts/pre-compact.sh +++ b/plugin/hooks/scripts/pre-compact.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash set -euo pipefail -# Prevent recursive hook firing from agent subshells -if [[ "${DEVSQUAD_HOOK_DEPTH:-0}" -ge 1 ]]; then +# Prevent recursive hook firing from durable workers and agent subshells. +if [[ "${DEVSQUAD_WORKER:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_DELEGATION_DEPTH:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_HOOK_DEPTH:-0}" != "0" ]]; then echo '{"hookSpecificOutput":{"hookEventName":"PreCompact"}}' exit 0 fi diff --git a/plugin/hooks/scripts/pre-tool-use.sh b/plugin/hooks/scripts/pre-tool-use.sh index c0c5d54..a7ea28c 100755 --- a/plugin/hooks/scripts/pre-tool-use.sh +++ b/plugin/hooks/scripts/pre-tool-use.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash set -euo pipefail -# Hook depth guard -- skip in agent subshells -if [[ "${DEVSQUAD_HOOK_DEPTH:-0}" -ge 1 ]]; then +# Recursion guard -- durable workers must not receive legacy delegation hooks. +if [[ "${DEVSQUAD_WORKER:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_DELEGATION_DEPTH:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_HOOK_DEPTH:-0}" != "0" ]]; then exit 0 fi diff --git a/plugin/hooks/scripts/session-start.sh b/plugin/hooks/scripts/session-start.sh index 4bd6045..1630a68 100755 --- a/plugin/hooks/scripts/session-start.sh +++ b/plugin/hooks/scripts/session-start.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash set -euo pipefail -# Prevent recursive hook firing from agent subshells -if [[ "${DEVSQUAD_HOOK_DEPTH:-0}" -ge 1 ]]; then +# Prevent recursive hook firing from durable workers and agent subshells. +if [[ "${DEVSQUAD_WORKER:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_DELEGATION_DEPTH:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_HOOK_DEPTH:-0}" != "0" ]]; then echo '{"hookSpecificOutput":{"hookEventName":"SessionStart","additionalContext":""}}' exit 0 fi diff --git a/plugin/hooks/scripts/stop.sh b/plugin/hooks/scripts/stop.sh index 46f2dac..54088d2 100755 --- a/plugin/hooks/scripts/stop.sh +++ b/plugin/hooks/scripts/stop.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash set -euo pipefail -# Hook depth guard -- skip in agent subshells -if [[ "${DEVSQUAD_HOOK_DEPTH:-0}" -ge 1 ]]; then +# Recursion guard -- durable workers must not receive legacy stop logic. +if [[ "${DEVSQUAD_WORKER:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_DELEGATION_DEPTH:-0}" != "0" ]] \ + || [[ "${DEVSQUAD_HOOK_DEPTH:-0}" != "0" ]]; then exit 0 fi diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index 04b94c3..89ace7d 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -51,7 +51,9 @@ def test_mcp_serve_dispatches_without_writing_protocol_stdout(self): with contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): code = cli.main(["mcp", "serve", "--runtime-dir", str(runtime)]) self.assertEqual((code, stdout.getvalue(), stderr.getvalue()), (0, "", "")) - serve.assert_called_once_with(runtime) + serve.assert_called_once_with( + runtime, caller_surface=None, caller_session_ref=None, + ) def test_missing_sdk_is_actionable_and_keeps_stdout_clean(self): stdout, stderr = io.StringIO(), io.StringIO() @@ -70,7 +72,9 @@ def setUp(self): self.runtime = Path(self.temp.name) / "runtime" (self.runtime / "artifacts").mkdir(parents=True) self.service = mock.Mock() - self.bridge = mcp_server.MCPBridge(self.runtime, self.service) + self.bridge = mcp_server.MCPBridge( + self.runtime, self.service, environment={}, + ) def tearDown(self): self.temp.cleanup() @@ -194,6 +198,76 @@ def test_result_rejects_preview_path_outside_the_runtime(self): self.assertEqual(payload["error"]["code"], "CONFLICT") self.assertIn("escapes", payload["error"]["message"]) + def test_configured_origin_is_saved_as_provenance_not_authorization(self): + task = {"schema_version": 1, "origin": {"surface": "user-label"}} + self.service.start.return_value = { + "run_id": "run-1", "state": "queued", "created": True, + } + bridge = mcp_server.MCPBridge( + self.runtime, + self.service, + caller_surface="codex-app", + caller_session_ref="thread-7", + environment={}, + ) + payload = bridge.start(task, "key-1") + self.assertTrue(payload["ok"]) + submitted = self.service.start.call_args.args[0] + self.assertEqual(submitted["origin"], { + "surface": "codex-app", "session_ref": "thread-7", + }) + self.assertEqual(task["origin"], {"surface": "user-label"}) + + self.service.reset_mock() + user_label_only = mcp_server.MCPBridge( + self.runtime, + self.service, + caller_surface="worker", + environment={}, + ) + self.service.start.return_value = { + "run_id": "run-2", "state": "queued", "created": True, + } + self.assertTrue(user_label_only.start(task, "key-2")["ok"]) + self.service.start.assert_called_once() + + def test_worker_environment_rejects_every_mutation_but_allows_inspection(self): + worker_bridge = mcp_server.MCPBridge( + self.runtime, + self.service, + environment={ + "DEVSQUAD_WORKER": "1", + "DEVSQUAD_RUN_ID": "run-1", + "DEVSQUAD_DELEGATION_DEPTH": "1", + }, + ) + mutations = [ + lambda: worker_bridge.start({"schema_version": 1}, "key-1"), + lambda: worker_bridge.cancel("run-1"), + lambda: worker_bridge.resume("run-1"), + lambda: worker_bridge.handoff_claim("run-1", 2, "host"), + lambda: worker_bridge.handoff_complete("run-1", {}, {}), + ] + for mutate in mutations: + with self.subTest(mutation=mutate): + payload = mutate() + self.assertFalse(payload["ok"]) + self.assertEqual(payload["error"]["code"], "POLICY_DENIED") + for method in ( + self.service.start, + self.service.cancel, + self.service.resume, + self.service.handoff_claim, + self.service.handoff_complete, + ): + method.assert_not_called() + + self.service.status.return_value = { + "run_id": "run-1", "state": "running", "version": 2, + } + self.assertTrue(worker_bridge.status("run-1")["ok"]) + self.service.status.assert_called_once_with("run-1") + @unittest.skipUnless(importlib.util.find_spec("mcp"), "optional MCP SDK is not installed") class OfficialSDKConformanceTest(unittest.TestCase): @@ -205,7 +279,9 @@ def test_tool_schemas_and_calls_use_the_official_in_memory_transport(self): service.start.return_value = {"run_id": "run-1", "state": "queued", "created": True} service.cancel.return_value = {"run_id": "run-1", "state": "cancelling", "version": 3} with tempfile.TemporaryDirectory(prefix="devsquad-sdk-server-") as directory: - server = mcp_server.build_server(Path(directory), service) + server = mcp_server.build_server( + Path(directory), service, environment={}, + ) async def probe(): async with Client(server) as client: diff --git a/test/test_hooks.sh b/test/test_hooks.sh index 385f0a5..74fbf3a 100644 --- a/test/test_hooks.sh +++ b/test/test_hooks.sh @@ -17,6 +17,8 @@ fresh_env() { export CLAUDE_PROJECT_DIR="$TEST_DIR" export HOME="$FAKE_HOME" unset DEVSQUAD_HOOK_DEPTH 2>/dev/null || true + unset DEVSQUAD_WORKER 2>/dev/null || true + unset DEVSQUAD_DELEGATION_DEPTH 2>/dev/null || true } run_hook() { @@ -44,6 +46,16 @@ assert_empty() { fi } +assert_path_missing() { + local label="$1" path="$2" + if [ ! -e "$path" ]; then + PASS=$((PASS + 1)) + else + FAIL=$((FAIL + 1)) + echo " FAIL: $label — unexpected path exists: $path" + fi +} + init_state() { bash -c "source '$PLUGIN_ROOT/lib/state.sh'; d=\$(init_state_dir); init_session_state \"\$d\"" >/dev/null } @@ -164,5 +176,23 @@ fresh_env init_state assert_contains "state dir self-ignores" "$(cat "$TEST_DIR/.devsquad/.gitignore" 2>/dev/null)" "*" +# --- Group 9: durable worker/delegation recursion guards --- +fresh_env +export DEVSQUAD_WORKER=1 +OUT=$(run_hook '{"tool_name":"WebSearch","tool_input":{"query":"nested"}}') +assert_empty "worker pre-tool hook is silent" "$OUT" +OUT=$(bash "$PLUGIN_ROOT/hooks/scripts/stop.sh" 2>/dev/null) +assert_empty "worker stop hook is silent" "$OUT" +OUT=$(bash "$PLUGIN_ROOT/hooks/scripts/session-start.sh" 2>/dev/null) +assert_contains "worker session-start returns valid empty context" "$OUT" '"additionalContext":""' +OUT=$(bash "$PLUGIN_ROOT/hooks/scripts/pre-compact.sh" 2>/dev/null) +assert_contains "worker pre-compact returns valid hook response" "$OUT" '"hookEventName":"PreCompact"' +assert_path_missing "worker hooks do not initialize plugin state" "$TEST_DIR/.devsquad" + +fresh_env +export DEVSQUAD_DELEGATION_DEPTH=2 +OUT=$(run_hook '{"tool_name":"WebSearch","tool_input":{"query":"nested-depth"}}') +assert_empty "delegated pre-tool hook is silent" "$OUT" + echo " hooks: ${PASS} passed, ${FAIL} failed" [ "$FAIL" -eq 0 ] From 4b784979ca2ad6ffc476ed38670c7143be5e229d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 22 Sep 2026 21:16:03 +0530 Subject: [PATCH 070/197] docs: checkpoint M4 MCP contract plan --- docs/plans/engineering-team/RESUME.md | 23 +++++++++++++++++------ docs/plans/engineering-team/backlog.json | 14 ++++++++++++-- 2 files changed, 29 insertions(+), 8 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 686ea09..58aaa50 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -41,6 +41,16 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne unknown capacity allows one unresolved trial, and later failure/cancellation retains earlier attempts and dispositions. A requested follow-up agent rerun hit the shared Plus limit; the 188-test complete gate is green after the fixes. +- M4 Plan 06-01 is complete at `643910d`. The optional official MCP Python + SDK is pinned and transitively locked at `mcp==2.2.0`; ordinary CLI and a + plain installed wheel remain dependency-free. `squad mcp serve` exposes the + eight saved-run operations with strict envelopes, bounded event pages and a + 16 KiB total artifact-preview cap. Worker-origin mutations are denied while + read-only inspection remains available, and all four legacy hooks honor the + worker/delegation guard. The gate is 200 core tests with `ResourceWarning` + promoted to failure, 208 Bash assertions, and 12 focused tests against the + installed official SDK. M4 remains in progress for setup/doctor and real + cross-surface receipts. - Last-observed provider readiness outside accepted M3: Claude CLI is not logged in; Grok CLI authentication expired; Gemini CLI's individual-account path is unsupported and its supported successor is Antigravity; Antigravity is @@ -67,8 +77,9 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 188 tests passed through M3 closeout with warnings promoted to errors | -| Bash 3.2 regression suite | 10 test files, 202 assertions passed | +| Python core discovery | 200 tests passed through M4 Plan 06-01 with warnings promoted to errors | +| Bash 3.2 regression suite | 10 test files, 208 assertions passed | +| Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 12 official-SDK focused tests passed | | Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 8 | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | | Integrated native adapter proof | Passed at `97a10f0`; gpt-5.5/low, read-only, correlated completion and confirmed process-group cleanup | @@ -100,10 +111,10 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Execute M4 from [IMPLEMENTATION.md](IMPLEMENTATION.md): pin/test the optional - MCP SDK without coupling it to core CLI imports, map the saved-run service - operations to strict stdio tools, enforce worker recursion guards, and add - idempotent local integration/doctor evidence. +2. Continue M4 Plan 06-02: add minimal host instructions and explicit local + stdio templates, implement idempotent registration/duplicate detection + while preserving unrelated settings, and make doctor report the executable + and arguments each installed app actually loads. 3. Keep Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. They do not block the independent Codex M3 gate. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index b60ed4d..6036de3 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -173,9 +173,19 @@ "id": "M4", "title": "Shared local MCP access and host handoffs", "depends_on": ["M3"], - "status": "pending", + "status": "in_progress", "acceptance_section": "M4 — Use that same run from local apps", - "evidence": [], + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "643910d", + "command_or_action": "200 core tests with ResourceWarning promoted to error, 208 shell assertions, Python 3.11 lock resolution and 12 focused official-SDK tests for the optional MCP boundary, saved-run tools and recursion guards", + "outcome": "M4 Plan 06-01 provides dependency-isolated stdio MCP access to all eight saved-run operations with strict envelopes, bounded previews, caller provenance and worker-origin mutation denial; setup/doctor and cross-surface live gates remain open", + "artifact": "../../../test/core/test_mcp.py", + "recorded_at": "2026-09-22T21:14:30+05:30", + "availability": "tracked_tests" + } + ], "blocker": null }, { From 96d990d153b7c5167de31558663ded2fd71438c4 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 22 Sep 2026 21:24:01 +0530 Subject: [PATCH 071/197] feat: define local MCP host templates --- .../engineering-team/MCP-LOCAL-ACCESS.md | 56 ++++++ .../antigravity/registration.json | 11 ++ .../claude-code/registration.json | 11 ++ .../core/integrations/codex/registration.json | 11 ++ .../core/integrations/grok/registration.json | 11 ++ plugin/core/pyproject.toml | 4 + plugin/core/src/devsquad/integrations.py | 164 ++++++++++++++++++ test/core/test_mcp.py | 76 ++++++++ 8 files changed, 344 insertions(+) create mode 100644 docs/plans/engineering-team/MCP-LOCAL-ACCESS.md create mode 100644 plugin/core/integrations/antigravity/registration.json create mode 100644 plugin/core/integrations/claude-code/registration.json create mode 100644 plugin/core/integrations/codex/registration.json create mode 100644 plugin/core/integrations/grok/registration.json create mode 100644 plugin/core/src/devsquad/integrations.py diff --git a/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md b/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md new file mode 100644 index 0000000..5fb2384 --- /dev/null +++ b/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md @@ -0,0 +1,56 @@ +# Local MCP access — operator contract + +This is the minimal M4 host contract for the shared local DevSquad runtime. It +does not create a cloud service, select a model for the host, or copy routing +policy into app-specific prompts. + +## One durable run, interchangeable clients + +Every supported local app launches the same absolute executable over stdio: + +```text +/absolute/path/to/squad mcp serve --surface HOST +``` + +The four versioned templates are under `plugin/core/integrations/`. Setup +resolves both the host CLI and `squad` to existing absolute files before it +registers anything. Host labels are saved as provenance only; they never grant +authorization. Worker/delegation environment markers, not a caller-supplied +label, deny recursive workflow mutations. + +Use the tools in this order: + +1. Construct one strict v1 task with bounded criteria, scope, checks and + budgets. Call `squad_start` once with a stable idempotency key and retain the + returned run ID. +2. Use `squad_status` for the projection and next action. Page + `squad_events` with `after=next_cursor`; do not treat provider reasoning as + an event stream. +3. Use `squad_result` for receipt/artifact IDs, paths and hashes. Its optional + UTF-8 previews share one 16 KiB ceiling; the files remain the authority. +4. If status reports `claim_handoff`, call `squad_handoff_claim` with the + displayed run version and a stable local owner label. Submit the returned + claim unchanged with a candidate-bound decision to + `squad_handoff_complete`. +5. `squad_cancel` saves cancellation intent; it does not wait for a worker's + lifetime. Use `squad_resume` only with the recovery object requested by + status. + +Closing an app or its MCP client does not cancel a saved run. The detached +worker is owned by the machine-local runtime, and another client can continue +with the same run ID. Two hosts still cannot advance the same handoff because +the service checks its fencing token and run version. + +## Registration templates + +| Host | CLI used by setup | Scope | Inspection form | +|---|---|---|---| +| Codex App/CLI | `codex mcp add` | Codex user config | `codex mcp get ... --json` | +| Claude Code | `claude mcp add --scope user` | user | `claude mcp get` | +| Antigravity | `agy mcp add` | Antigravity user config | `agy mcp list` | +| Grok Build | `grok mcp add --scope user` | user | `grok mcp list --json` | + +The next M4 slice implements duplicate-aware `squad setup` and doctor output +over these templates. Until that checkpoint, the templates are tested package +data and a registration specification, not a claim that any host is already +configured or connected. diff --git a/plugin/core/integrations/antigravity/registration.json b/plugin/core/integrations/antigravity/registration.json new file mode 100644 index 0000000..bf3ded4 --- /dev/null +++ b/plugin/core/integrations/antigravity/registration.json @@ -0,0 +1,11 @@ +{ + "schema_version": 1, + "id": "antigravity", + "display_name": "Antigravity", + "executable_names": ["agy", "antigravity"], + "server_name": "devsquad", + "surface": "antigravity", + "register_argv": ["{host_executable}", "mcp", "add", "{server_name}", "--", "{squad_executable}", "mcp", "serve", "--surface", "{surface}"], + "inspect_argv": ["{host_executable}", "mcp", "list"], + "inspect_format": "text" +} diff --git a/plugin/core/integrations/claude-code/registration.json b/plugin/core/integrations/claude-code/registration.json new file mode 100644 index 0000000..ee21f55 --- /dev/null +++ b/plugin/core/integrations/claude-code/registration.json @@ -0,0 +1,11 @@ +{ + "schema_version": 1, + "id": "claude-code", + "display_name": "Claude Code", + "executable_names": ["claude"], + "server_name": "devsquad", + "surface": "claude-code", + "register_argv": ["{host_executable}", "mcp", "add", "--scope", "user", "{server_name}", "--", "{squad_executable}", "mcp", "serve", "--surface", "{surface}"], + "inspect_argv": ["{host_executable}", "mcp", "get", "{server_name}"], + "inspect_format": "text" +} diff --git a/plugin/core/integrations/codex/registration.json b/plugin/core/integrations/codex/registration.json new file mode 100644 index 0000000..adc0fb8 --- /dev/null +++ b/plugin/core/integrations/codex/registration.json @@ -0,0 +1,11 @@ +{ + "schema_version": 1, + "id": "codex", + "display_name": "Codex", + "executable_names": ["codex"], + "server_name": "devsquad", + "surface": "codex-app", + "register_argv": ["{host_executable}", "mcp", "add", "{server_name}", "--", "{squad_executable}", "mcp", "serve", "--surface", "{surface}"], + "inspect_argv": ["{host_executable}", "mcp", "get", "{server_name}", "--json"], + "inspect_format": "json" +} diff --git a/plugin/core/integrations/grok/registration.json b/plugin/core/integrations/grok/registration.json new file mode 100644 index 0000000..b354ea2 --- /dev/null +++ b/plugin/core/integrations/grok/registration.json @@ -0,0 +1,11 @@ +{ + "schema_version": 1, + "id": "grok", + "display_name": "Grok Build", + "executable_names": ["grok"], + "server_name": "devsquad", + "surface": "grok-build", + "register_argv": ["{host_executable}", "mcp", "add", "--scope", "user", "{server_name}", "--", "{squad_executable}", "mcp", "serve", "--surface", "{surface}"], + "inspect_argv": ["{host_executable}", "mcp", "list", "--json"], + "inspect_format": "json" +} diff --git a/plugin/core/pyproject.toml b/plugin/core/pyproject.toml index b28e8cb..b469d18 100644 --- a/plugin/core/pyproject.toml +++ b/plugin/core/pyproject.toml @@ -30,3 +30,7 @@ where = ["src"] "share/devsquad/adapters" = ["adapters/classification-policy.conf"] "share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/check-result.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/profiles.schema.json", "schemas/review-result.schema.json", "schemas/task.schema.json"] "share/devsquad/profiles" = ["profiles/templates.json"] +"share/devsquad/integrations/codex" = ["integrations/codex/registration.json"] +"share/devsquad/integrations/claude-code" = ["integrations/claude-code/registration.json"] +"share/devsquad/integrations/antigravity" = ["integrations/antigravity/registration.json"] +"share/devsquad/integrations/grok" = ["integrations/grok/registration.json"] diff --git a/plugin/core/src/devsquad/integrations.py b/plugin/core/src/devsquad/integrations.py new file mode 100644 index 0000000..e10b50b --- /dev/null +++ b/plugin/core/src/devsquad/integrations.py @@ -0,0 +1,164 @@ +"""Strict local MCP host-registration templates.""" + +from __future__ import annotations + +from dataclasses import dataclass +import json +import os +from pathlib import Path +import string +import sys +from typing import Any + +from .contracts import ContractError + +SOURCE_ROOT = Path(__file__).resolve().parents[2] +CORE_ROOT = ( + SOURCE_ROOT + if (SOURCE_ROOT / "integrations").is_dir() + else Path(sys.prefix) / "share" / "devsquad" +) +TEMPLATE_FIELDS = { + "schema_version", + "id", + "display_name", + "executable_names", + "server_name", + "surface", + "register_argv", + "inspect_argv", + "inspect_format", +} +PLACEHOLDERS = { + "host_executable", "squad_executable", "server_name", "surface", +} + + +def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ContractError(f"duplicate integration template key: {key}") + result[key] = value + return result + + +@dataclass(frozen=True) +class IntegrationTemplate: + id: str + display_name: str + executable_names: tuple[str, ...] + server_name: str + surface: str + register_argv: tuple[str, ...] + inspect_argv: tuple[str, ...] + inspect_format: str + path: Path + + @classmethod + def load(cls, path: Path) -> "IntegrationTemplate": + try: + value = json.loads(path.read_text(), object_pairs_hook=_unique_object) + except (OSError, UnicodeError, json.JSONDecodeError) as exc: + raise ContractError(f"cannot read integration template {path}: {exc}") from exc + if not isinstance(value, dict) or set(value) != TEMPLATE_FIELDS: + fields = set(value) if isinstance(value, dict) else set() + raise ContractError( + f"integration template fields differ: {sorted(fields ^ TEMPLATE_FIELDS)}" + ) + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("unsupported integration template schema_version") + for field in ("id", "display_name", "server_name", "surface"): + if not isinstance(value[field], str) or not value[field]: + raise ContractError(f"integration {field} must be non-empty") + for field in ("executable_names", "register_argv", "inspect_argv"): + sequence = value[field] + if (not isinstance(sequence, list) or not sequence + or not all(isinstance(item, str) and item for item in sequence)): + raise ContractError(f"integration {field} must be a non-empty string array") + if field == "executable_names" and len(set(sequence)) != len(sequence): + raise ContractError("integration executable_names must be unique") + if value["inspect_format"] not in {"json", "text"}: + raise ContractError("integration inspect_format is invalid") + cls._validate_placeholders(value["register_argv"]) + cls._validate_placeholders(value["inspect_argv"]) + return cls( + id=value["id"], + display_name=value["display_name"], + executable_names=tuple(value["executable_names"]), + server_name=value["server_name"], + surface=value["surface"], + register_argv=tuple(value["register_argv"]), + inspect_argv=tuple(value["inspect_argv"]), + inspect_format=value["inspect_format"], + path=path, + ) + + @staticmethod + def _validate_placeholders(argv: list[str]) -> None: + formatter = string.Formatter() + referenced = { + name + for argument in argv + for _, name, _, _ in formatter.parse(argument) + if name is not None + } + if referenced - PLACEHOLDERS: + raise ContractError( + f"unknown integration placeholders: {sorted(referenced - PLACEHOLDERS)}" + ) + + def _render( + self, + argv: tuple[str, ...], + *, + host_executable: Path, + squad_executable: Path, + ) -> tuple[str, ...]: + try: + host = host_executable.resolve(strict=True) + squad = squad_executable.resolve(strict=True) + except OSError as exc: + raise ContractError("integration executables must exist") from exc + if (not host.is_file() or not squad.is_file() + or not os.access(host, os.X_OK) or not os.access(squad, os.X_OK)): + raise ContractError("integration executables must be executable files") + values = { + "host_executable": str(host), + "squad_executable": str(squad), + "server_name": self.server_name, + "surface": self.surface, + } + return tuple(argument.format_map(values) for argument in argv) + + def registration_command( + self, host_executable: Path, squad_executable: Path, + ) -> tuple[str, ...]: + return self._render( + self.register_argv, + host_executable=host_executable, + squad_executable=squad_executable, + ) + + def inspection_command( + self, host_executable: Path, squad_executable: Path, + ) -> tuple[str, ...]: + return self._render( + self.inspect_argv, + host_executable=host_executable, + squad_executable=squad_executable, + ) + + +def load_integrations(root: Path | None = None) -> tuple[IntegrationTemplate, ...]: + integration_root = root or (CORE_ROOT / "integrations") + templates = tuple( + IntegrationTemplate.load(path) + for path in sorted(integration_root.glob("*/registration.json")) + ) + if not templates: + raise ContractError("no local MCP integration templates are installed") + ids = [template.id for template in templates] + if len(set(ids)) != len(ids): + raise ContractError("integration template ids must be unique") + return templates diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index 89ace7d..51f9c1a 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -2,6 +2,7 @@ import asyncio import io import importlib.util +import json import os from pathlib import Path import shutil @@ -18,6 +19,7 @@ from devsquad import cli, mcp_server from devsquad.contracts import ContractError +from devsquad.integrations import IntegrationTemplate, load_integrations from devsquad.store import ConflictError @@ -66,6 +68,64 @@ def test_missing_sdk_is_actionable_and_keeps_stdout_clean(self): self.assertIn("devsquad-core[mcp]", stderr.getvalue()) +class MCPIntegrationTemplateTest(unittest.TestCase): + def test_four_host_templates_render_absolute_argv_without_a_shell(self): + templates = {template.id: template for template in load_integrations()} + self.assertEqual(set(templates), { + "codex", "claude-code", "antigravity", "grok", + }) + host = Path(sys.executable) + squad = CORE / "bin/squad" + prefixes = { + "codex": ("mcp", "add", "devsquad", "--"), + "claude-code": ( + "mcp", "add", "--scope", "user", "devsquad", "--", + ), + "antigravity": ("mcp", "add", "devsquad", "--"), + "grok": ( + "mcp", "add", "--scope", "user", "devsquad", "--", + ), + } + for integration_id, template in templates.items(): + with self.subTest(integration=integration_id): + command = template.registration_command(host, squad) + resolved_host = str(host.resolve(strict=True)) + resolved_squad = str(squad.resolve(strict=True)) + self.assertEqual(command[0], resolved_host) + self.assertEqual(command[1:1 + len(prefixes[integration_id])], prefixes[integration_id]) + squad_index = command.index(resolved_squad) + self.assertEqual(command[squad_index + 1:squad_index + 4], ( + "mcp", "serve", "--surface", + )) + self.assertEqual(command[squad_index + 4], template.surface) + self.assertFalse(any("{" in argument for argument in command)) + inspection = template.inspection_command(host, squad) + self.assertEqual(inspection[0], resolved_host) + self.assertIn("mcp", inspection) + + def test_template_schema_rejects_unknown_placeholders_and_fields(self): + with tempfile.TemporaryDirectory(prefix="devsquad-template-") as directory: + path = Path(directory) / "registration.json" + template = { + "schema_version": 1, + "id": "bad", + "display_name": "Bad", + "executable_names": ["bad"], + "server_name": "devsquad", + "surface": "bad", + "register_argv": ["{unknown}"], + "inspect_argv": ["{host_executable}"], + "inspect_format": "text", + } + path.write_text(json.dumps(template)) + with self.assertRaisesRegex(ContractError, "unknown integration placeholders"): + IntegrationTemplate.load(path) + template["unexpected"] = True + path.write_text(json.dumps(template)) + with self.assertRaisesRegex(ContractError, "fields differ"): + IntegrationTemplate.load(path) + + class MCPBridgeTest(unittest.TestCase): def setUp(self): self.temp = tempfile.TemporaryDirectory(prefix="devsquad-mcp-bridge-") @@ -372,6 +432,22 @@ def test_plain_installed_wheel_keeps_cli_usable_without_mcp(self): [str(squad), "--version"], check=True, text=True, capture_output=True, env=environment, ) self.assertEqual(version.stdout.strip(), "squad 0.1.0") + integrations = subprocess.run( + [ + str(python), "-P", "-c", + "from devsquad.integrations import load_integrations; " + "print(','.join(sorted(item.id for item in load_integrations())))", + ], + check=True, + text=True, + capture_output=True, + cwd=root, + env=environment, + ) + self.assertEqual( + integrations.stdout.strip(), + "antigravity,claude-code,codex,grok", + ) missing = subprocess.run( [str(squad), "mcp", "serve"], text=True, capture_output=True, env=environment, ) From 9abad1ef99f9f5d7ef3b5d2b283de06d5ddeb1b7 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 04:21:39 +0530 Subject: [PATCH 072/197] feat: add idempotent local MCP setup --- .../engineering-team/MCP-LOCAL-ACCESS.md | 43 +- .../antigravity/registration.json | 1 + .../claude-code/registration.json | 1 + .../core/integrations/codex/registration.json | 1 + .../core/integrations/grok/registration.json | 1 + plugin/core/src/devsquad/cli.py | 63 ++- plugin/core/src/devsquad/diagnostics.py | 92 +++ plugin/core/src/devsquad/integrations.py | 527 +++++++++++++++++- plugin/core/src/devsquad/mcp_server.py | 16 + test/core/test_cli.py | 65 +++ test/core/test_mcp.py | 317 ++++++++++- 11 files changed, 1101 insertions(+), 26 deletions(-) create mode 100644 plugin/core/src/devsquad/diagnostics.py diff --git a/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md b/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md index 5fb2384..6662784 100644 --- a/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md +++ b/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md @@ -50,7 +50,42 @@ the service checks its fencing token and run version. | Antigravity | `agy mcp add` | Antigravity user config | `agy mcp list` | | Grok Build | `grok mcp add --scope user` | user | `grok mcp list --json` | -The next M4 slice implements duplicate-aware `squad setup` and doctor output -over these templates. Until that checkpoint, the templates are tested package -data and a registration specification, not a claim that any host is already -configured or connected. +Install the optional, exactly pinned MCP extra into the same stable environment +that owns the `squad` executable, then preview or apply registration: + +```text +python3 -m pip install './plugin/core[mcp]' +squad setup --dry-run --json +squad setup --json +squad doctor --json +``` + +Use `--host codex`, `--host claude-code`, `--host antigravity` or +`--host grok` to limit setup; repeat `--host` for more than one. Setup calls +the installed host CLI with argument arrays, never a shell command, and then +re-inspects what that host reports as loaded. A second successful setup is a +no-op. Claude Code requires a targeted user-scope remove/add only when its +existing direct `devsquad` entry has drifted because its CLI does not replace a +named server in place. + +Setup fails closed instead of editing an ambiguous configuration when it sees: + +- the server in more than one direct scope; +- a loaded registration that differs from the one visible in the direct + config, indicating an inherited override; +- a project/local registration, malformed config or a config the host does + not load; +- an unresolved launcher, unavailable host CLI, missing SDK or any MCP SDK + version other than the supported `2.2.0` pin. + +`squad doctor --json` is read-only. It reports adapter versions, the resolved +launcher, installed SDK version, config paths, normalized host inspection and +whether each installed app loads the expected absolute command. Environment +maps are never returned. Arguments are returned only when they exactly match +the fixed DevSquad server arguments; drifted arguments are replaced by a count +and SHA-256 digest so credentials cannot be echoed. Doctor exits 1 when an +installed app is not ready, while unavailable apps are not treated as required. + +These commands prove local CLI registration and SDK conformance. Actual +in-app operation still requires the cross-surface receipts in M4/M7; config +syntax or a matching `mcp list` result is not presented as that live proof. diff --git a/plugin/core/integrations/antigravity/registration.json b/plugin/core/integrations/antigravity/registration.json index bf3ded4..d07622e 100644 --- a/plugin/core/integrations/antigravity/registration.json +++ b/plugin/core/integrations/antigravity/registration.json @@ -6,6 +6,7 @@ "server_name": "devsquad", "surface": "antigravity", "register_argv": ["{host_executable}", "mcp", "add", "{server_name}", "--", "{squad_executable}", "mcp", "serve", "--surface", "{surface}"], + "remove_argv": null, "inspect_argv": ["{host_executable}", "mcp", "list"], "inspect_format": "text" } diff --git a/plugin/core/integrations/claude-code/registration.json b/plugin/core/integrations/claude-code/registration.json index ee21f55..5558857 100644 --- a/plugin/core/integrations/claude-code/registration.json +++ b/plugin/core/integrations/claude-code/registration.json @@ -6,6 +6,7 @@ "server_name": "devsquad", "surface": "claude-code", "register_argv": ["{host_executable}", "mcp", "add", "--scope", "user", "{server_name}", "--", "{squad_executable}", "mcp", "serve", "--surface", "{surface}"], + "remove_argv": ["{host_executable}", "mcp", "remove", "--scope", "user", "{server_name}"], "inspect_argv": ["{host_executable}", "mcp", "get", "{server_name}"], "inspect_format": "text" } diff --git a/plugin/core/integrations/codex/registration.json b/plugin/core/integrations/codex/registration.json index adc0fb8..90744d2 100644 --- a/plugin/core/integrations/codex/registration.json +++ b/plugin/core/integrations/codex/registration.json @@ -6,6 +6,7 @@ "server_name": "devsquad", "surface": "codex-app", "register_argv": ["{host_executable}", "mcp", "add", "{server_name}", "--", "{squad_executable}", "mcp", "serve", "--surface", "{surface}"], + "remove_argv": null, "inspect_argv": ["{host_executable}", "mcp", "get", "{server_name}", "--json"], "inspect_format": "json" } diff --git a/plugin/core/integrations/grok/registration.json b/plugin/core/integrations/grok/registration.json index b354ea2..dbcd462 100644 --- a/plugin/core/integrations/grok/registration.json +++ b/plugin/core/integrations/grok/registration.json @@ -6,6 +6,7 @@ "server_name": "devsquad", "surface": "grok-build", "register_argv": ["{host_executable}", "mcp", "add", "--scope", "user", "{server_name}", "--", "{squad_executable}", "mcp", "serve", "--surface", "{surface}"], + "remove_argv": null, "inspect_argv": ["{host_executable}", "mcp", "list", "--json"], "inspect_format": "json" } diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 4f6cc2a..4002064 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -13,6 +13,8 @@ from . import __version__ from .adapters import AdapterManifest, classify_cli, harness_version, prepare_cli, prepare_native_codex_from_catalog from .contracts import ContractError, envelope, error_payload +from .diagnostics import build_doctor_report +from .integrations import LocalIntegrationManager, load_integrations from .service import Service from .store import ConflictError, SchemaVersionError @@ -33,15 +35,41 @@ def manifests() -> list[tuple[Path, AdapterManifest]]: return [(path, AdapterManifest.load(path)) for path in sorted((CORE_ROOT / "adapters").glob("*/adapter.json"))] -def command_doctor(_: argparse.Namespace) -> tuple[dict, int]: - rows = [] - for path, manifest in manifests(): - binary = manifest.resolve_binary() - version = harness_version(binary) if binary else None - status = "unavailable" if not binary else ("supported" if version in manifest.verified_versions else "unverified") - rows.append({"adapter": manifest.name, "transport": manifest.transport, "status": status, "binary": binary, "version": version, "manifest": str(path)}) - ready = any(row["status"] != "unavailable" for row in rows) - return envelope(data={"core_version": __version__, "ready": ready, "adapters": rows}), 0 if ready else 1 +def command_doctor(args: argparse.Namespace) -> tuple[dict, int]: + report = build_doctor_report( + project=Path(args.project_dir), + squad_executable=( + Path(args.squad_executable) if args.squad_executable else None + ), + ) + return envelope(data=report), 0 if report["ready"] else 1 + + +def command_setup(args: argparse.Namespace) -> tuple[dict, int]: + templates = load_integrations() + selected = set(args.host or (template.id for template in templates)) + manager = LocalIntegrationManager( + project=Path(args.project_dir), + squad_executable=( + Path(args.squad_executable) if args.squad_executable else None + ), + ) + rows = [ + manager.setup(template, dry_run=args.dry_run) + for template in templates + if template.id in selected + ] + successful_actions = { + "added", "updated", "unchanged", "would_add", "would_update", + } + completed = all(row["action"] in successful_actions for row in rows) + ready = all(row["ready"] for row in rows) + return envelope(data={ + "completed": completed, + "ready": ready, + "dry_run": args.dry_run, + "hosts": rows, + }), 0 if completed else 1 def _read_json(path: str, label: str) -> Any: @@ -165,7 +193,22 @@ def parser() -> argparse.ArgumentParser: p = ContractParser(prog="squad") p.add_argument("--version", action="version", version=f"squad {__version__}") sub = p.add_subparsers(dest="command", required=True) - doctor = sub.add_parser("doctor"); doctor.add_argument("--json", action="store_true"); doctor.set_defaults(func=command_doctor) + doctor = sub.add_parser("doctor") + doctor.add_argument("--json", action="store_true") + doctor.add_argument("--project-dir", default=str(Path.cwd())) + doctor.add_argument("--squad-executable") + doctor.set_defaults(func=command_doctor) + setup = sub.add_parser("setup") + setup.add_argument( + "--host", + action="append", + choices=("codex", "claude-code", "antigravity", "grok"), + ) + setup.add_argument("--dry-run", action="store_true") + setup.add_argument("--json", action="store_true") + setup.add_argument("--project-dir", default=str(Path.cwd())) + setup.add_argument("--squad-executable") + setup.set_defaults(func=command_setup) for name, fn in (("prepare", command_prepare), ("classify", command_classify)): cmd = sub.add_parser(name) cmd.add_argument("adapter", choices=("codex", "antigravity", "grok")) diff --git a/plugin/core/src/devsquad/diagnostics.py b/plugin/core/src/devsquad/diagnostics.py new file mode 100644 index 0000000..a58edf0 --- /dev/null +++ b/plugin/core/src/devsquad/diagnostics.py @@ -0,0 +1,92 @@ +"""Read-only readiness reporting shared by CLI and MCP surfaces.""" + +from __future__ import annotations + +from pathlib import Path +import sys +from typing import Any + +from . import __version__ +from .adapters import AdapterManifest, harness_version +from .integrations import ( + LocalIntegrationManager, + load_integrations, +) + +SOURCE_ROOT = Path(__file__).resolve().parents[2] +CORE_ROOT = ( + SOURCE_ROOT + if (SOURCE_ROOT / "adapters").is_dir() + else Path(sys.prefix) / "share" / "devsquad" +) + + +def _adapter_rows() -> list[dict[str, Any]]: + rows = [] + for path in sorted((CORE_ROOT / "adapters").glob("*/adapter.json")): + manifest = AdapterManifest.load(path) + binary = manifest.resolve_binary() + version = harness_version(binary) if binary else None + status = ( + "unavailable" if not binary + else "supported" if version in manifest.verified_versions + else "unverified" + ) + rows.append({ + "adapter": manifest.name, + "transport": manifest.transport, + "status": status, + "binary": binary, + "version": version, + "manifest": str(path), + }) + return rows + + +def build_doctor_report( + *, + project: Path, + home: Path | None = None, + squad_executable: Path | None = None, + manager: LocalIntegrationManager | None = None, +) -> dict[str, Any]: + """Report provider and installed-app readiness without modifying config.""" + + adapters = _adapter_rows() + integration_manager = manager or LocalIntegrationManager( + project=project, + home=home, + squad_executable=squad_executable, + ) + local_apps = [ + integration_manager.inspect(template) for template in load_integrations() + ] + installed_apps = [row for row in local_apps if row["installed"]] + adapter_ready = any(row["status"] != "unavailable" for row in adapters) + local_apps_required = bool(installed_apps) + local_apps_ready = ( + all(row["ready"] for row in installed_apps) + if local_apps_required else True + ) + return { + "core_version": __version__, + "ready": adapter_ready and local_apps_ready, + "adapter_ready": adapter_ready, + "local_app_access": { + "required": local_apps_required, + "ready": local_apps_ready, + "installed_count": len(installed_apps), + "configured_count": sum(row["ready"] for row in installed_apps), + "mcp_sdk_requirement": "mcp==2.2.0", + "mcp_sdk_available": integration_manager.mcp_sdk_available, + "mcp_sdk_supported": integration_manager.mcp_sdk_supported, + "mcp_sdk_version": integration_manager.mcp_sdk_version, + "squad_executable": ( + str(integration_manager.squad_executable) + if integration_manager.squad_executable else None + ), + "launcher_error": integration_manager.launcher_error, + }, + "adapters": adapters, + "local_apps": local_apps, + } diff --git a/plugin/core/src/devsquad/integrations.py b/plugin/core/src/devsquad/integrations.py index e10b50b..d804667 100644 --- a/plugin/core/src/devsquad/integrations.py +++ b/plugin/core/src/devsquad/integrations.py @@ -3,12 +3,20 @@ from __future__ import annotations from dataclasses import dataclass +from importlib import metadata +import importlib.util +import hashlib import json import os from pathlib import Path +import re +import shlex +import shutil import string +import subprocess import sys -from typing import Any +import tomllib +from typing import Any, Callable from .contracts import ContractError @@ -26,6 +34,7 @@ "server_name", "surface", "register_argv", + "remove_argv", "inspect_argv", "inspect_format", } @@ -51,6 +60,7 @@ class IntegrationTemplate: server_name: str surface: str register_argv: tuple[str, ...] + remove_argv: tuple[str, ...] | None inspect_argv: tuple[str, ...] inspect_format: str path: Path @@ -78,10 +88,17 @@ def load(cls, path: Path) -> "IntegrationTemplate": raise ContractError(f"integration {field} must be a non-empty string array") if field == "executable_names" and len(set(sequence)) != len(sequence): raise ContractError("integration executable_names must be unique") + remove_argv = value["remove_argv"] + if (remove_argv is not None + and (not isinstance(remove_argv, list) or not remove_argv + or not all(isinstance(item, str) and item for item in remove_argv))): + raise ContractError("integration remove_argv must be null or a non-empty string array") if value["inspect_format"] not in {"json", "text"}: raise ContractError("integration inspect_format is invalid") cls._validate_placeholders(value["register_argv"]) cls._validate_placeholders(value["inspect_argv"]) + if remove_argv is not None: + cls._validate_placeholders(remove_argv) return cls( id=value["id"], display_name=value["display_name"], @@ -89,6 +106,7 @@ def load(cls, path: Path) -> "IntegrationTemplate": server_name=value["server_name"], surface=value["surface"], register_argv=tuple(value["register_argv"]), + remove_argv=tuple(remove_argv) if remove_argv is not None else None, inspect_argv=tuple(value["inspect_argv"]), inspect_format=value["inspect_format"], path=path, @@ -113,19 +131,33 @@ def _render( argv: tuple[str, ...], *, host_executable: Path, - squad_executable: Path, + squad_executable: Path | None, ) -> tuple[str, ...]: try: host = host_executable.resolve(strict=True) - squad = squad_executable.resolve(strict=True) except OSError as exc: - raise ContractError("integration executables must exist") from exc - if (not host.is_file() or not squad.is_file() - or not os.access(host, os.X_OK) or not os.access(squad, os.X_OK)): - raise ContractError("integration executables must be executable files") + raise ContractError("host executable must exist") from exc + if not host.is_file() or not os.access(host, os.X_OK): + raise ContractError("host executable must be an executable file") + referenced = { + name + for argument in argv + for _, name, _, _ in string.Formatter().parse(argument) + if name is not None + } + squad = None + if "squad_executable" in referenced: + if squad_executable is None: + raise ContractError("squad executable is not resolved") + try: + squad = squad_executable.resolve(strict=True) + except OSError as exc: + raise ContractError("squad executable must exist") from exc + if not squad.is_file() or not os.access(squad, os.X_OK): + raise ContractError("squad executable must be an executable file") values = { "host_executable": str(host), - "squad_executable": str(squad), + "squad_executable": str(squad) if squad else "", "server_name": self.server_name, "surface": self.surface, } @@ -141,7 +173,7 @@ def registration_command( ) def inspection_command( - self, host_executable: Path, squad_executable: Path, + self, host_executable: Path, squad_executable: Path | None = None, ) -> tuple[str, ...]: return self._render( self.inspect_argv, @@ -149,6 +181,15 @@ def inspection_command( squad_executable=squad_executable, ) + def removal_command(self, host_executable: Path) -> tuple[str, ...] | None: + if self.remove_argv is None: + return None + return self._render( + self.remove_argv, + host_executable=host_executable, + squad_executable=None, + ) + def load_integrations(root: Path | None = None) -> tuple[IntegrationTemplate, ...]: integration_root = root or (CORE_ROOT / "integrations") @@ -162,3 +203,471 @@ def load_integrations(root: Path | None = None) -> tuple[IntegrationTemplate, .. if len(set(ids)) != len(ids): raise ContractError("integration template ids must be unique") return templates + + +def resolve_squad_executable(explicit: Path | None = None) -> Path: + """Resolve one executable that remains valid outside the current shell.""" + + candidates = ( + [explicit] + if explicit is not None + else [ + Path(found) if (found := shutil.which("squad")) else None, + CORE_ROOT / "bin" / "squad", + ] + ) + for candidate in candidates: + if candidate is None: + continue + try: + resolved = candidate.resolve(strict=True) + except OSError: + continue + if resolved.is_file() and os.access(resolved, os.X_OK): + return resolved + raise ContractError( + "cannot resolve a stable squad executable; install devsquad-core or " + "pass --squad-executable" + ) + + +def _registration( + *, scope: str, path: Path, value: Any, disabled_key: str | None = None, +) -> dict[str, Any]: + valid = isinstance(value, dict) + command = value.get("command") if valid else None + args = value.get("args", []) if valid else None + if not isinstance(command, str) or not command: + valid = False + if not isinstance(args, list) or not all(isinstance(item, str) for item in args): + valid = False + enabled = True + if isinstance(value, dict): + if disabled_key is not None: + enabled = value.get(disabled_key) is not True + elif "enabled" in value: + enabled = value.get("enabled") is True + return { + "scope": scope, + "path": str(path), + "command": command if isinstance(command, str) else None, + "args": args if isinstance(args, list) else None, + "enabled": enabled, + "valid": valid, + } + + +def _safe_parse_error(path: Path, format_name: str) -> str: + return f"cannot parse {format_name} configuration at {path}" + + +def _json_document(path: Path) -> Any: + return json.loads(path.read_text(), object_pairs_hook=_unique_object) + + +def _toml_document(path: Path) -> Any: + return tomllib.loads(path.read_text()) + + +def registration_sources( + template: IntegrationTemplate, + *, + home: Path, + project: Path, +) -> tuple[list[dict[str, Any]], list[dict[str, str]]]: + """Find direct and inherited registrations without returning env/secrets.""" + + project = project.resolve() + candidates: list[tuple[str, Path, str, tuple[str, ...], str | None]] = [] + if template.id == "codex": + candidates = [ + ("user", home / ".codex/config.toml", "toml", ("mcp_servers",), None), + ("project", project / ".codex/config.toml", "toml", ("mcp_servers",), None), + ] + elif template.id == "grok": + candidates = [ + ("user", home / ".grok/config.toml", "toml", ("mcp_servers",), None), + ("project", project / ".grok/config.toml", "toml", ("mcp_servers",), None), + ] + elif template.id == "antigravity": + candidates = [ + ("user", home / ".gemini/config/mcp_config.json", "json", ("mcpServers",), "disabled"), + ("project", project / ".gemini/config/mcp_config.json", "json", ("mcpServers",), "disabled"), + ("project", project / ".gemini/settings.json", "json", ("mcpServers",), "disabled"), + ] + elif template.id == "claude-code": + candidates = [ + ("user", home / ".claude.json", "json", ("mcpServers",), None), + ("project", project / ".mcp.json", "json", ("mcpServers",), None), + ( + "local", + home / ".claude.json", + "json", + ("projects", str(project), "mcpServers"), + None, + ), + ] + registrations: list[dict[str, Any]] = [] + errors: list[dict[str, str]] = [] + documents: dict[tuple[Path, str], Any] = {} + invalid_documents: set[tuple[Path, str]] = set() + for scope, path, format_name, key_path, disabled_key in candidates: + if not path.is_file(): + continue + cache_key = (path, format_name) + if cache_key in invalid_documents: + continue + if cache_key not in documents: + try: + documents[cache_key] = ( + _json_document(path) if format_name == "json" else _toml_document(path) + ) + except (OSError, UnicodeError, json.JSONDecodeError, tomllib.TOMLDecodeError, ContractError): + invalid_documents.add(cache_key) + errors.append({ + "scope": scope, + "path": str(path), + "error": _safe_parse_error(path, format_name), + }) + continue + value = documents[cache_key] + for key in key_path: + if not isinstance(value, dict) or key not in value: + value = None + break + value = value[key] + if isinstance(value, dict) and template.server_name in value: + registrations.append(_registration( + scope=scope, + path=path, + value=value[template.server_name], + disabled_key=disabled_key, + )) + return registrations, errors + + +def _actual_from_json(template: IntegrationTemplate, stdout: str) -> dict[str, Any] | None: + try: + value = json.loads(stdout) + except json.JSONDecodeError: + return None + if template.id == "codex" and isinstance(value, dict): + transport = value.get("transport") + if value.get("name") == template.server_name and isinstance(transport, dict): + return { + "command": transport.get("command"), + "args": transport.get("args"), + "enabled": value.get("enabled") is True, + "scope": None, + } + if template.id == "grok" and isinstance(value, list): + for item in value: + if isinstance(item, dict) and item.get("name") == template.server_name: + return { + "command": item.get("command"), + "args": item.get("args"), + "enabled": item.get("enabled") is True, + "scope": item.get("scope"), + } + return None + + +def _actual_from_text(template: IntegrationTemplate, stdout: str) -> dict[str, Any] | None: + if template.id == "claude-code": + fields: dict[str, str] = {} + for line in stdout.splitlines(): + if ":" in line: + key, value = line.split(":", 1) + fields[key.strip().lower()] = value.strip() + if template.server_name + ":" not in stdout.lower() or "command" not in fields: + return None + try: + args = shlex.split(fields.get("args", "")) + except ValueError: + return None + return { + "command": fields["command"], + "args": args, + "enabled": not fields.get("status", "").lower().startswith("disabled"), + "scope": fields.get("scope"), + "connection": ( + "failed" if "failed" in fields.get("status", "").lower() + else "connected" if "connected" in fields.get("status", "").lower() + else "unknown" + ), + } + if template.id == "antigravity": + for line in stdout.splitlines(): + columns = re.split(r"\s{2,}", line.strip(), maxsplit=3) + if len(columns) == 4 and columns[0] == template.server_name: + try: + command = shlex.split(columns[3]) + except ValueError: + return None + if not command: + return None + expected_args = ["mcp", "serve", "--surface", template.surface] + if len(command) > len(expected_args) and command[-len(expected_args):] == expected_args: + executable = " ".join(command[:-len(expected_args)]) + arguments = expected_args + else: + executable = command[0] + arguments = command[1:] + return { + "command": executable, + "args": arguments, + "enabled": columns[2].lower() == "enabled", + "scope": None, + } + return None + + +def _sdk_version() -> str | None: + if importlib.util.find_spec("mcp") is None: + return None + try: + return metadata.version("mcp") + except metadata.PackageNotFoundError: + return None + + +def _redacted_registration( + registration: dict[str, Any], expected: dict[str, Any], +) -> dict[str, Any]: + """Expose matching argv, but never echo arbitrary drifted arguments.""" + + arguments = registration.get("args") + args_match = arguments == expected["args"] + public = { + key: value + for key, value in registration.items() + if key not in {"args"} + } + public["args"] = arguments if args_match else None + public["args_match"] = args_match + if isinstance(arguments, list): + canonical = json.dumps(arguments, ensure_ascii=True, separators=(",", ":")) + public["args_sha256"] = hashlib.sha256(canonical.encode()).hexdigest() + public["args_count"] = len(arguments) + else: + public["args_sha256"] = None + public["args_count"] = None + return public + + +class LocalIntegrationManager: + """Inspect and idempotently register the local stdio server via host CLIs.""" + + def __init__( + self, + *, + project: Path, + home: Path | None = None, + squad_executable: Path | None = None, + which: Callable[[str], str | None] = shutil.which, + runner: Callable[..., subprocess.CompletedProcess[str]] = subprocess.run, + timeout_seconds: float = 8, + mcp_sdk_available: bool | None = None, + mcp_sdk_version: str | None = None, + ): + self.project = project.resolve(strict=True) + self.home = (home or Path.home()).resolve(strict=True) + self.launcher_error = None + try: + self.squad_executable = resolve_squad_executable(squad_executable) + except ContractError as exc: + self.squad_executable = None + self.launcher_error = str(exc) + self.which = which + self.runner = runner + self.timeout_seconds = timeout_seconds + detected_version = _sdk_version() + self.mcp_sdk_version = ( + detected_version if mcp_sdk_version is None else mcp_sdk_version + ) + self.mcp_sdk_available = ( + detected_version is not None + if mcp_sdk_available is None else mcp_sdk_available + ) + self.mcp_sdk_supported = ( + self.mcp_sdk_available and self.mcp_sdk_version == "2.2.0" + ) + + def _host_executable(self, template: IntegrationTemplate) -> Path | None: + for name in template.executable_names: + found = self.which(name) + if not found: + continue + try: + resolved = Path(found).resolve(strict=True) + except OSError: + continue + if resolved.is_file() and os.access(resolved, os.X_OK): + return resolved + return None + + def _run(self, command: tuple[str, ...]) -> subprocess.CompletedProcess[str] | None: + environment = os.environ.copy() + environment["HOME"] = str(self.home) + try: + return self.runner( + command, + cwd=self.project, + env=environment, + text=True, + capture_output=True, + timeout=self.timeout_seconds, + ) + except (OSError, subprocess.TimeoutExpired): + return None + + def inspect(self, template: IntegrationTemplate) -> dict[str, Any]: + host = self._host_executable(template) + expected = { + "command": str(self.squad_executable) if self.squad_executable else None, + "args": ["mcp", "serve", "--surface", template.surface], + } + sources, source_errors = registration_sources( + template, home=self.home, project=self.project, + ) + actual = None + inspection = {"completed": False, "exit_code": None} + if host is not None: + command = template.inspection_command(host) + completed = self._run(command) + if completed is not None: + inspection = { + "completed": True, + "exit_code": completed.returncode, + } + if completed.returncode == 0: + actual = ( + _actual_from_json(template, completed.stdout) + if template.inspect_format == "json" + else _actual_from_text(template, completed.stdout) + ) + matches = bool( + actual + and actual.get("command") == expected["command"] + and actual.get("args") == expected["args"] + and actual.get("enabled") is True + ) + loaded_matches_source = bool( + actual + and len(sources) == 1 + and actual.get("command") == sources[0].get("command") + and actual.get("args") == sources[0].get("args") + and actual.get("enabled") == sources[0].get("enabled") + ) + if host is None: + status = "unavailable" + elif source_errors: + status = "invalid_config" + elif len(sources) > 1: + status = "duplicate" + elif any(not source["valid"] for source in sources): + status = "invalid_config" + elif self.squad_executable is None: + status = "unstable_launcher" + elif actual is None and sources: + status = "not_loaded" + elif actual is None: + status = "missing" + elif not sources: + status = "inherited" + elif not loaded_matches_source: + status = "duplicate" + elif matches: + status = "matching" + else: + status = "drifted" + public_sources = [ + _redacted_registration(source, expected) for source in sources + ] + public_actual = ( + _redacted_registration(actual, expected) if actual is not None else None + ) + return { + "id": template.id, + "display_name": template.display_name, + "installed": host is not None, + "host_executable": str(host) if host else None, + "server_name": template.server_name, + "status": status, + "ready": status == "matching" and self.mcp_sdk_supported, + "mcp_sdk_available": self.mcp_sdk_available, + "mcp_sdk_supported": self.mcp_sdk_supported, + "mcp_sdk_version": self.mcp_sdk_version, + "expected": expected, + "loaded": public_actual, + "sources": public_sources, + "source_errors": source_errors, + "inspection": inspection, + "template": str(template.path), + "launcher_error": self.launcher_error, + } + + def setup(self, template: IntegrationTemplate, *, dry_run: bool = False) -> dict[str, Any]: + before = self.inspect(template) + status = before["status"] + if not self.mcp_sdk_available: + return {**before, "action": "blocked_missing_mcp_sdk", "changed": False} + if not self.mcp_sdk_supported: + return {**before, "action": "blocked_unsupported_mcp_sdk", "changed": False} + if status == "unavailable": + return {**before, "action": "blocked_host_unavailable", "changed": False} + if status == "unstable_launcher": + return {**before, "action": "blocked_unstable_launcher", "changed": False} + if status == "matching": + return {**before, "action": "unchanged", "changed": False} + if status in {"duplicate", "inherited", "not_loaded", "invalid_config"}: + return {**before, "action": f"blocked_{status}", "changed": False} + if status == "drifted" and before["sources"][0]["scope"] != "user": + return {**before, "action": "blocked_inherited", "changed": False} + action = "add" if status == "missing" else "update" + if dry_run: + return {**before, "action": f"would_{action}", "changed": False} + host = Path(before["host_executable"]) + removed = False + removal_command = template.removal_command(host) if action == "update" else None + if removal_command is not None: + removal = self._run(removal_command) + if removal is None or removal.returncode != 0: + return { + **before, + "action": "update_failed", + "changed": None, + "removal_exit_code": ( + removal.returncode if removal is not None else None + ), + } + removed = True + completed = self._run( + template.registration_command(host, self.squad_executable) + ) + if completed is None or completed.returncode != 0: + return { + **before, + "action": f"{action}_failed", + "changed": True if removed else None, + "removal_exit_code": 0 if removed else None, + "registration_exit_code": ( + completed.returncode if completed is not None else None + ), + } + after = self.inspect(template) + if not after["ready"]: + return { + **after, + "action": f"{action}_incomplete", + "changed": True, + "removal_exit_code": 0 if removed else None, + "registration_exit_code": completed.returncode, + } + return { + **after, + "action": "added" if action == "add" else "updated", + "changed": True, + "removal_exit_code": 0 if removed else None, + "registration_exit_code": completed.returncode, + } diff --git a/plugin/core/src/devsquad/mcp_server.py b/plugin/core/src/devsquad/mcp_server.py index 9516f15..679d278 100644 --- a/plugin/core/src/devsquad/mcp_server.py +++ b/plugin/core/src/devsquad/mcp_server.py @@ -13,6 +13,7 @@ from . import __version__ from .contracts import ContractError, PolicyDenied, envelope, error_payload +from .diagnostics import build_doctor_report from .service import Service from .store import ConflictError @@ -49,6 +50,7 @@ def __init__( caller_surface: str | None = None, caller_session_ref: str | None = None, environment: Mapping[str, str] | None = None, + project: Path | None = None, ): for label, value in ( ("caller_surface", caller_surface), @@ -61,6 +63,7 @@ def __init__( self.caller_surface = caller_surface self.caller_session_ref = caller_session_ref self.environment = os.environ if environment is None else environment + self.project = (project or Path.cwd()).resolve() @staticmethod def _response(operation: Callable[[], dict[str, Any]]) -> dict[str, Any]: @@ -118,6 +121,11 @@ def start( ) ) + def doctor(self) -> dict[str, Any]: + """Inspect local provider and application readiness without mutation.""" + + return self._response(lambda: build_doctor_report(project=self.project)) + def status(self, run_id: str) -> dict[str, Any]: """Inspect the current projection for a saved run.""" @@ -221,6 +229,7 @@ def build_server( caller_surface: str | None = None, caller_session_ref: str | None = None, environment: Mapping[str, str] | None = None, + project: Path | None = None, ) -> Any: """Build the local stdio server without starting it.""" @@ -231,6 +240,7 @@ def build_server( caller_surface=caller_surface, caller_session_ref=caller_session_ref, environment=environment, + project=project, ) server = server_type( "DevSquad", @@ -241,6 +251,12 @@ def build_server( ), ) + @server.tool(name="squad_doctor") + def squad_doctor() -> dict[str, Any]: + """Inspect versions, capabilities and local host registration drift.""" + + return bridge.doctor() + @server.tool(name="squad_start") def squad_start( task: dict[str, Any], diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 92ea72c..fb8fa77 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -179,6 +179,71 @@ def test_parser_and_json_file_failures_are_input_errors(self): self.assertEqual(payload["error"]["code"], "INPUT_INVALID") self.assertIn("cannot read task file", payload["error"]["message"]) + def test_setup_filters_hosts_and_treats_a_valid_dry_run_as_completed(self): + codex = mock.Mock(id="codex") + grok = mock.Mock(id="grok") + manager = mock.Mock() + manager.setup.return_value = { + "id": "codex", + "status": "missing", + "ready": False, + "action": "would_add", + "changed": False, + } + squad = ROOT / "plugin/core/bin/squad" + with ( + mock.patch.object(cli, "load_integrations", return_value=(codex, grok)), + mock.patch.object(cli, "LocalIntegrationManager", return_value=manager) as constructor, + ): + code, payload, stderr = self.invoke([ + "setup", "--host", "codex", "--dry-run", + "--project-dir", str(self.root), + "--squad-executable", str(squad), "--json", + ]) + self.assertEqual((code, stderr), (0, "")) + self.assertTrue(payload["data"]["completed"]) + self.assertFalse(payload["data"]["ready"]) + self.assertTrue(payload["data"]["dry_run"]) + constructor.assert_called_once_with( + project=self.root, squad_executable=squad, + ) + manager.setup.assert_called_once_with(codex, dry_run=True) + + def test_setup_and_doctor_return_one_when_readiness_is_blocked(self): + template = mock.Mock(id="codex") + manager = mock.Mock() + manager.setup.return_value = { + "id": "codex", + "status": "duplicate", + "ready": False, + "action": "blocked_duplicate", + "changed": False, + } + with ( + mock.patch.object(cli, "load_integrations", return_value=(template,)), + mock.patch.object(cli, "LocalIntegrationManager", return_value=manager), + ): + code, payload, stderr = self.invoke([ + "setup", "--project-dir", str(self.root), "--json", + ]) + self.assertEqual((code, stderr), (1, "")) + self.assertFalse(payload["data"]["completed"]) + self.assertEqual(payload["data"]["hosts"][0]["action"], "blocked_duplicate") + + report = { + "core_version": "0.1.0", "ready": False, + "adapters": [], "local_apps": [], + } + with mock.patch.object(cli, "build_doctor_report", return_value=report) as doctor: + code, payload, stderr = self.invoke([ + "doctor", "--project-dir", str(self.root), "--json", + ]) + self.assertEqual((code, stderr), (1, "")) + self.assertEqual(payload["data"], report) + doctor.assert_called_once_with( + project=self.root, squad_executable=None, + ) + def test_contract_conflict_schema_and_runtime_errors_have_exact_exits(self): cases = [ (ContractError("bad input"), 64, "INPUT_INVALID"), diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index 51f9c1a..ccb59e7 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -5,6 +5,7 @@ import json import os from pathlib import Path +import shlex import shutil import subprocess import sys @@ -17,9 +18,13 @@ CORE = ROOT / "plugin/core" sys.path.insert(0, str(CORE / "src")) -from devsquad import cli, mcp_server +from devsquad import cli, diagnostics, mcp_server from devsquad.contracts import ContractError -from devsquad.integrations import IntegrationTemplate, load_integrations +from devsquad.integrations import ( + IntegrationTemplate, + LocalIntegrationManager, + load_integrations, +) from devsquad.store import ConflictError @@ -102,6 +107,12 @@ def test_four_host_templates_render_absolute_argv_without_a_shell(self): inspection = template.inspection_command(host, squad) self.assertEqual(inspection[0], resolved_host) self.assertIn("mcp", inspection) + removal = template.removal_command(host) + if integration_id == "claude-code": + self.assertIsNotNone(removal) + self.assertIn("remove", removal) + else: + self.assertIsNone(removal) def test_template_schema_rejects_unknown_placeholders_and_fields(self): with tempfile.TemporaryDirectory(prefix="devsquad-template-") as directory: @@ -114,6 +125,7 @@ def test_template_schema_rejects_unknown_placeholders_and_fields(self): "server_name": "devsquad", "surface": "bad", "register_argv": ["{unknown}"], + "remove_argv": None, "inspect_argv": ["{host_executable}"], "inspect_format": "text", } @@ -126,6 +138,299 @@ def test_template_schema_rejects_unknown_placeholders_and_fields(self): IntegrationTemplate.load(path) +class FakeMCPHost: + def __init__(self, template, home): + self.template = template + self.home = home + self.loaded = None + self.registration_calls = 0 + + def _config_path(self): + return { + "codex": self.home / ".codex/config.toml", + "claude-code": self.home / ".claude.json", + "antigravity": self.home / ".gemini/config/mcp_config.json", + "grok": self.home / ".grok/config.toml", + }[self.template.id] + + def seed_unrelated_config(self): + path = self._config_path() + path.parent.mkdir(parents=True, exist_ok=True) + if path.suffix == ".json": + path.write_text(json.dumps({"unrelated": {"credential": "preserve-me"}})) + else: + path.write_text('unrelated = "preserve-me"\n') + + def _save_registration(self, command, args): + path = self._config_path() + path.parent.mkdir(parents=True, exist_ok=True) + if path.suffix == ".json": + value = json.loads(path.read_text()) if path.exists() else {} + value.setdefault("mcpServers", {})["devsquad"] = { + "command": command, + "args": args, + } + if self.template.id == "antigravity": + value["mcpServers"]["devsquad"]["disabled"] = False + path.write_text(json.dumps(value)) + else: + existing = path.read_text() if path.exists() else "" + if "[mcp_servers.devsquad]" not in existing: + enabled = "enabled = true\n" if self.template.id == "grok" else "" + path.write_text( + existing + + "\n[mcp_servers.devsquad]\n" + + f"command = {json.dumps(command)}\n" + + f"args = {json.dumps(args)}\n" + + enabled + ) + + def _inspection_result(self, argv): + if self.loaded is None: + if self.template.id == "grok": + return subprocess.CompletedProcess(argv, 0, "[]\n", "") + if self.template.id == "antigravity": + return subprocess.CompletedProcess( + argv, 0, "NAME TYPE STATUS COMMAND/URL\n", "", + ) + return subprocess.CompletedProcess(argv, 1, "", "not found") + command, args = self.loaded + if self.template.id == "codex": + stdout = json.dumps({ + "name": "devsquad", + "enabled": True, + "transport": {"command": command, "args": args, "env": None}, + }) + elif self.template.id == "grok": + stdout = json.dumps([{ + "name": "devsquad", "enabled": True, "scope": "user", + "command": command, "args": args, + }]) + elif self.template.id == "claude-code": + stdout = ( + "devsquad:\n" + " Scope: User config (available in all your projects)\n" + " Status: ✓ Connected\n" + " Type: stdio\n" + f" Command: {command}\n" + f" Args: {shlex.join(args)}\n" + " Environment: PRIVATE_TOKEN=not-reported\n" + ) + else: + stdout = ( + "NAME TYPE STATUS COMMAND/URL\n" + f"devsquad stdio enabled {shlex.join([command, *args])}\n" + ) + return subprocess.CompletedProcess(argv, 0, stdout, "") + + def __call__(self, argv, **_): + argv = tuple(argv) + if "remove" in argv: + self.loaded = None + path = self._config_path() + value = json.loads(path.read_text()) + value.get("mcpServers", {}).pop("devsquad", None) + path.write_text(json.dumps(value)) + return subprocess.CompletedProcess(argv, 0, "removed\n", "") + if "add" not in argv: + return self._inspection_result(argv) + self.registration_calls += 1 + delimiter = argv.index("--") + command = argv[delimiter + 1] + args = list(argv[delimiter + 2:]) + self.loaded = (command, args) + self._save_registration(command, args) + return subprocess.CompletedProcess(argv, 0, "registered\n", "") + + +class LocalMCPRegistrationTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-local-mcp-") + self.root = Path(self.temp.name) + self.home = self.root / "home" + self.project = self.root / "project" + self.home.mkdir() + self.project.mkdir() + self.squad = CORE / "bin/squad" + + def tearDown(self): + self.temp.cleanup() + + def manager(self, fake): + return LocalIntegrationManager( + project=self.project, + home=self.home, + squad_executable=self.squad, + which=lambda _: sys.executable, + runner=fake, + mcp_sdk_available=True, + mcp_sdk_version="2.2.0", + ) + + def test_setup_is_idempotent_for_every_host_and_preserves_unrelated_config(self): + for template in load_integrations(): + with self.subTest(host=template.id): + fake = FakeMCPHost(template, self.home) + fake.seed_unrelated_config() + manager = self.manager(fake) + + first = manager.setup(template) + second = manager.setup(template) + + self.assertEqual(first["action"], "added") + self.assertTrue(first["ready"]) + self.assertEqual(second["action"], "unchanged") + self.assertTrue(second["ready"]) + self.assertEqual(fake.registration_calls, 1) + self.assertEqual(len(second["sources"]), 1) + self.assertEqual(second["loaded"]["args"], [ + "mcp", "serve", "--surface", template.surface, + ]) + self.assertNotIn("PRIVATE_TOKEN", json.dumps(second)) + self.assertIn("preserve-me", fake._config_path().read_text()) + + fake._config_path().unlink() + + def test_duplicate_and_inherited_registrations_fail_closed_without_mutation(self): + template = next(item for item in load_integrations() if item.id == "codex") + expected = ( + str(self.squad.resolve()), + ["mcp", "serve", "--surface", template.surface], + ) + fake = FakeMCPHost(template, self.home) + fake.loaded = expected + fake._save_registration(*expected) + project_config = self.project / ".codex/config.toml" + project_config.parent.mkdir(parents=True) + project_config.write_text( + "[mcp_servers.devsquad]\n" + f"command = {json.dumps(expected[0])}\n" + f"args = {json.dumps(expected[1])}\n" + ) + duplicate = self.manager(fake).setup(template) + self.assertEqual(duplicate["status"], "duplicate") + self.assertEqual(duplicate["action"], "blocked_duplicate") + self.assertEqual(fake.registration_calls, 0) + fake._config_path().unlink() + project_config.unlink() + inherited = self.manager(fake).setup(template) + self.assertEqual(inherited["status"], "inherited") + self.assertEqual(inherited["action"], "blocked_inherited") + self.assertEqual(fake.registration_calls, 0) + + fake._save_registration(*expected) + fake.loaded = ("/inherited/override", ["mcp", "serve"]) + overlaid = self.manager(fake).setup(template) + self.assertEqual(overlaid["status"], "duplicate") + self.assertEqual(overlaid["action"], "blocked_duplicate") + self.assertEqual(fake.registration_calls, 0) + + def test_doctor_data_redacts_drifted_arguments_and_malformed_config_content(self): + template = next( + item for item in load_integrations() if item.id == "claude-code" + ) + fake = FakeMCPHost(template, self.home) + fake.loaded = (str(self.squad.resolve()), ["--api-key", "super-secret"]) + fake._save_registration(*fake.loaded) + drifted = self.manager(fake).inspect(template) + encoded = json.dumps(drifted) + self.assertEqual(drifted["status"], "drifted") + self.assertIsNone(drifted["loaded"]["args"]) + self.assertIsNone(drifted["sources"][0]["args"]) + self.assertNotIn("super-secret", encoded) + + updated = self.manager(fake).setup(template) + self.assertEqual(updated["action"], "updated") + self.assertTrue(updated["ready"]) + self.assertEqual(updated["removal_exit_code"], 0) + self.assertEqual(fake.registration_calls, 1) + + fake._config_path().write_text('{"private":"do-not-report"') + fake.loaded = None + malformed = self.manager(fake).inspect(template) + self.assertEqual(malformed["status"], "invalid_config") + self.assertNotIn("do-not-report", json.dumps(malformed)) + + def test_setup_requires_the_exact_supported_optional_sdk(self): + template = next(item for item in load_integrations() if item.id == "grok") + fake = FakeMCPHost(template, self.home) + missing = LocalIntegrationManager( + project=self.project, + home=self.home, + squad_executable=self.squad, + which=lambda _: sys.executable, + runner=fake, + mcp_sdk_available=False, + ).setup(template) + self.assertEqual(missing["action"], "blocked_missing_mcp_sdk") + unsupported = LocalIntegrationManager( + project=self.project, + home=self.home, + squad_executable=self.squad, + which=lambda _: sys.executable, + runner=fake, + mcp_sdk_available=True, + mcp_sdk_version="2.1.0", + ).setup(template) + self.assertEqual(unsupported["action"], "blocked_unsupported_mcp_sdk") + self.assertEqual(fake.registration_calls, 0) + + def test_explicit_missing_launcher_does_not_silently_fall_back(self): + template = next(item for item in load_integrations() if item.id == "codex") + fake = FakeMCPHost(template, self.home) + manager = LocalIntegrationManager( + project=self.project, + home=self.home, + squad_executable=self.root / "missing-squad", + which=lambda _: sys.executable, + runner=fake, + mcp_sdk_available=True, + mcp_sdk_version="2.2.0", + ) + result = manager.setup(template) + self.assertEqual(result["status"], "unstable_launcher") + self.assertEqual(result["action"], "blocked_unstable_launcher") + self.assertIsNone(result["expected"]["command"]) + self.assertEqual(fake.registration_calls, 0) + + +class MCPDoctorReportTest(unittest.TestCase): + def test_installed_app_drift_controls_readiness_but_unavailable_apps_do_not(self): + manager = mock.Mock( + mcp_sdk_available=True, + mcp_sdk_supported=True, + mcp_sdk_version="2.2.0", + squad_executable=CORE / "bin/squad", + launcher_error=None, + ) + rows = [ + {"id": "codex", "installed": True, "ready": True}, + {"id": "claude-code", "installed": False, "ready": False}, + ] + manager.inspect.side_effect = rows + templates = (mock.Mock(id="codex"), mock.Mock(id="claude-code")) + adapters = [{"adapter": "codex", "status": "supported"}] + with ( + mock.patch.object(diagnostics, "_adapter_rows", return_value=adapters), + mock.patch.object(diagnostics, "load_integrations", return_value=templates), + ): + report = diagnostics.build_doctor_report(project=ROOT, manager=manager) + self.assertTrue(report["ready"]) + self.assertEqual(report["local_app_access"]["installed_count"], 1) + self.assertEqual(report["local_app_access"]["configured_count"], 1) + + manager.inspect.side_effect = [ + {"id": "codex", "installed": True, "ready": False}, rows[1], + ] + with ( + mock.patch.object(diagnostics, "_adapter_rows", return_value=adapters), + mock.patch.object(diagnostics, "load_integrations", return_value=templates), + ): + drifted = diagnostics.build_doctor_report(project=ROOT, manager=manager) + self.assertFalse(drifted["ready"]) + self.assertFalse(drifted["local_app_access"]["ready"]) + + class MCPBridgeTest(unittest.TestCase): def setUp(self): self.temp = tempfile.TemporaryDirectory(prefix="devsquad-mcp-bridge-") @@ -193,6 +498,12 @@ def test_operations_map_directly_to_the_saved_run_service(self): method.assert_called_once_with(*expected_args) method.reset_mock() + def test_doctor_uses_the_shared_read_only_report(self): + report = {"core_version": "0.1.0", "ready": True, "local_apps": []} + with mock.patch.object(mcp_server, "build_doctor_report", return_value=report) as doctor: + self.assert_success(self.bridge.doctor(), report) + doctor.assert_called_once_with(project=Path.cwd().resolve()) + def test_contract_conflict_and_internal_failures_keep_machine_envelopes(self): cases = [ (ContractError("bad request"), "INPUT_INVALID"), @@ -348,7 +659,7 @@ async def probe(): listing = await client.list_tools() tools = {tool.name: tool for tool in listing.tools} self.assertEqual(set(tools), { - "squad_start", "squad_status", "squad_events", "squad_result", + "squad_doctor", "squad_start", "squad_status", "squad_events", "squad_result", "squad_cancel", "squad_resume", "squad_handoff_claim", "squad_handoff_complete", }) From 56641132db1d574ea8b47f1e1f92c962a52ed0ff Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 04:24:12 +0530 Subject: [PATCH 073/197] docs: checkpoint M4 local setup plan --- docs/plans/engineering-team/RESUME.md | 30 +++++++++++++++++------- docs/plans/engineering-team/backlog.json | 9 +++++++ 2 files changed, 30 insertions(+), 9 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 58aaa50..bffacb1 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,7 +2,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 22, 2026 +## Current position — September 23, 2026 - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` remains the published runtime @@ -49,8 +49,19 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne read-only inspection remains available, and all four legacy hooks honor the worker/delegation guard. The gate is 200 core tests with `ResourceWarning` promoted to failure, 208 Bash assertions, and 12 focused tests against the - installed official SDK. M4 remains in progress for setup/doctor and real - cross-surface receipts. + installed official SDK. +- M4 Plan 06-02 is complete at `9abad1e`. Four packaged host templates now + drive duplicate-aware `squad setup`; `squad doctor` and the ninth MCP tool, + `squad_doctor`, report the app-loaded command, SDK and registration drift + without returning environment maps or arbitrary arguments. Registration is + fail-closed for inherited/duplicate/malformed config and is idempotent while + preserving unrelated settings. The gate is 211 core tests with + `ResourceWarning` promoted to failure, 208 Bash assertions and 21 focused + tests under the pinned SDK. Installed Codex 0.135.0, Claude 2.1.220, + Antigravity 1.2.3 and Grok 0.2.111 CLIs each passed add, second-run no-op and + drift-repair checks under an isolated temporary HOME. That is CLI/config + evidence, not a claim of in-app operation; real account configs remain + untouched. M4 remains in progress for Plan 06-03 cross-surface receipts. - Last-observed provider readiness outside accepted M3: Claude CLI is not logged in; Grok CLI authentication expired; Gemini CLI's individual-account path is unsupported and its supported successor is Antigravity; Antigravity is @@ -77,9 +88,10 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 200 tests passed through M4 Plan 06-01 with warnings promoted to errors | +| Python core discovery | 211 tests passed through M4 Plan 06-02 with warnings promoted to errors | | Bash 3.2 regression suite | 10 test files, 208 assertions passed | -| Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 12 official-SDK focused tests passed | +| Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 21 official-SDK focused tests passed | +| M4 local host setup | Four installed host CLIs registered under an isolated HOME, repaired drift and made no second-run changes; duplicate/inherited configs fail closed | | Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 8 | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | | Integrated native adapter proof | Passed at `97a10f0`; gpt-5.5/low, read-only, correlated completion and confirmed process-group cleanup | @@ -111,10 +123,10 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M4 Plan 06-02: add minimal host instructions and explicit local - stdio templates, implement idempotent registration/duplicate detection - while preserving unrelated settings, and make doctor report the executable - and arguments each installed app actually loads. +2. Execute M4 Plan 06-03: install one stable isolated local MCP runtime, use + the idempotent setup path for the available host surfaces, and prove one + saved run across terminal and Codex with client-disconnect survival, + handoff fencing, duplicate detection and installed-version receipts. 3. Keep Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. They do not block the independent Codex M3 gate. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 6036de3..05cade9 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -184,6 +184,15 @@ "artifact": "../../../test/core/test_mcp.py", "recorded_at": "2026-09-22T21:14:30+05:30", "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "9abad1e", + "command_or_action": "211 core tests with ResourceWarning promoted to error, 208 shell assertions, 21 pinned-SDK tests and isolated-HOME add/no-op/drift-repair checks through the installed Codex, Claude, Antigravity and Grok host CLIs", + "outcome": "M4 Plan 06-02 provides strict host templates, stable launcher resolution, idempotent setup, inherited/duplicate fail-closed detection, redacted installed-app doctor reporting and the squad_doctor MCP tool; real in-app cross-surface receipts remain open", + "artifact": "MCP-LOCAL-ACCESS.md", + "recorded_at": "2026-09-23T04:22:26+05:30", + "availability": "tracked_tests" } ], "blocker": null From 5b82f537fe91c20017d58f7f8ef8e9d238a1cf8d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 04:30:57 +0530 Subject: [PATCH 074/197] fix: prefer app-bundled Codex host --- .../antigravity/registration.json | 1 + .../claude-code/registration.json | 1 + .../core/integrations/codex/registration.json | 1 + .../core/integrations/grok/registration.json | 1 + plugin/core/src/devsquad/integrations.py | 31 +++++++++++++------ test/core/test_mcp.py | 1 + 6 files changed, 27 insertions(+), 9 deletions(-) diff --git a/plugin/core/integrations/antigravity/registration.json b/plugin/core/integrations/antigravity/registration.json index d07622e..7764b89 100644 --- a/plugin/core/integrations/antigravity/registration.json +++ b/plugin/core/integrations/antigravity/registration.json @@ -2,6 +2,7 @@ "schema_version": 1, "id": "antigravity", "display_name": "Antigravity", + "executable_paths": [], "executable_names": ["agy", "antigravity"], "server_name": "devsquad", "surface": "antigravity", diff --git a/plugin/core/integrations/claude-code/registration.json b/plugin/core/integrations/claude-code/registration.json index 5558857..119a32f 100644 --- a/plugin/core/integrations/claude-code/registration.json +++ b/plugin/core/integrations/claude-code/registration.json @@ -2,6 +2,7 @@ "schema_version": 1, "id": "claude-code", "display_name": "Claude Code", + "executable_paths": [], "executable_names": ["claude"], "server_name": "devsquad", "surface": "claude-code", diff --git a/plugin/core/integrations/codex/registration.json b/plugin/core/integrations/codex/registration.json index 90744d2..1372b3b 100644 --- a/plugin/core/integrations/codex/registration.json +++ b/plugin/core/integrations/codex/registration.json @@ -2,6 +2,7 @@ "schema_version": 1, "id": "codex", "display_name": "Codex", + "executable_paths": ["/Applications/ChatGPT.app/Contents/Resources/codex", "/Applications/Codex.app/Contents/Resources/codex"], "executable_names": ["codex"], "server_name": "devsquad", "surface": "codex-app", diff --git a/plugin/core/integrations/grok/registration.json b/plugin/core/integrations/grok/registration.json index dbcd462..d06932e 100644 --- a/plugin/core/integrations/grok/registration.json +++ b/plugin/core/integrations/grok/registration.json @@ -2,6 +2,7 @@ "schema_version": 1, "id": "grok", "display_name": "Grok Build", + "executable_paths": [], "executable_names": ["grok"], "server_name": "devsquad", "surface": "grok-build", diff --git a/plugin/core/src/devsquad/integrations.py b/plugin/core/src/devsquad/integrations.py index d804667..630a34d 100644 --- a/plugin/core/src/devsquad/integrations.py +++ b/plugin/core/src/devsquad/integrations.py @@ -30,6 +30,7 @@ "schema_version", "id", "display_name", + "executable_paths", "executable_names", "server_name", "surface", @@ -56,6 +57,7 @@ def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: class IntegrationTemplate: id: str display_name: str + executable_paths: tuple[str, ...] executable_names: tuple[str, ...] server_name: str surface: str @@ -81,13 +83,20 @@ def load(cls, path: Path) -> "IntegrationTemplate": for field in ("id", "display_name", "server_name", "surface"): if not isinstance(value[field], str) or not value[field]: raise ContractError(f"integration {field} must be non-empty") - for field in ("executable_names", "register_argv", "inspect_argv"): + for field in ( + "executable_paths", "executable_names", "register_argv", "inspect_argv", + ): sequence = value[field] - if (not isinstance(sequence, list) or not sequence + allow_empty = field == "executable_paths" + if (not isinstance(sequence, list) or (not allow_empty and not sequence) or not all(isinstance(item, str) and item for item in sequence)): - raise ContractError(f"integration {field} must be a non-empty string array") - if field == "executable_names" and len(set(sequence)) != len(sequence): - raise ContractError("integration executable_names must be unique") + qualifier = "a string array" if allow_empty else "a non-empty string array" + raise ContractError(f"integration {field} must be {qualifier}") + if field in {"executable_paths", "executable_names"}: + if len(set(sequence)) != len(sequence): + raise ContractError(f"integration {field} must be unique") + if not all(Path(item).is_absolute() for item in value["executable_paths"]): + raise ContractError("integration executable_paths must be absolute") remove_argv = value["remove_argv"] if (remove_argv is not None and (not isinstance(remove_argv, list) or not remove_argv @@ -102,6 +111,7 @@ def load(cls, path: Path) -> "IntegrationTemplate": return cls( id=value["id"], display_name=value["display_name"], + executable_paths=tuple(value["executable_paths"]), executable_names=tuple(value["executable_names"]), server_name=value["server_name"], surface=value["surface"], @@ -494,10 +504,13 @@ def __init__( ) def _host_executable(self, template: IntegrationTemplate) -> Path | None: - for name in template.executable_names: - found = self.which(name) - if not found: - continue + candidates: list[str] = list(template.executable_paths) + candidates.extend( + found + for name in template.executable_names + if (found := self.which(name)) is not None + ) + for found in candidates: try: resolved = Path(found).resolve(strict=True) except OSError: diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index ccb59e7..1c259fe 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -121,6 +121,7 @@ def test_template_schema_rejects_unknown_placeholders_and_fields(self): "schema_version": 1, "id": "bad", "display_name": "Bad", + "executable_paths": [], "executable_names": ["bad"], "server_name": "devsquad", "surface": "bad", From b7cd626ade5367a7cb12ceab221698982914f6b0 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 06:50:09 +0530 Subject: [PATCH 075/197] test: prove MCP client disconnect survival --- test/core/test_mcp.py | 75 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 75 insertions(+) diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index 1c259fe..82cd849 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -10,6 +10,7 @@ import subprocess import sys import tempfile +import time import tomllib import unittest from unittest import mock @@ -25,7 +26,9 @@ LocalIntegrationManager, load_integrations, ) +from devsquad.service import Service from devsquad.store import ConflictError +from devsquad_test_fixtures import branch_review_routing_documents class MCPDependencyBoundaryTest(unittest.TestCase): @@ -685,6 +688,78 @@ async def probe(): asyncio.run(probe()) + def test_closing_a_real_stdio_client_does_not_cancel_the_detached_worker(self): + from mcp import Client, StdioServerParameters + + with tempfile.TemporaryDirectory(prefix="devsquad-sdk-disconnect-") as directory: + root = Path(directory) + repo = root / "repo" + runtime = root / "runtime" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + subprocess.run( + ["git", "-C", str(repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(repo), "config", "user.name", "Test"], + check=True, + ) + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "src/app.py").write_text("VALUE = 'fixture'\n") + (repo / "tests/test_app.py").write_text("# fixture\n") + profiles, policy = branch_review_routing_documents() + (repo / "profiles.json").write_text(profiles) + (repo / "policy.json").write_text(policy) + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run( + ["git", "-C", str(repo), "commit", "-qm", "fixture"], + check=True, + ) + task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + task["project"] = { + "repo_path": str(repo), "base_ref": "HEAD", "target_ref": "HEAD", + } + task["routing"] = { + "profiles_file": "profiles.json", "policy_file": "policy.json", + } + service = Service(runtime) + started = service.start( + task, "stdio-client-disconnect", _internal_fake_delay=2, + ) + + async def observe_then_disconnect(): + parameters = StdioServerParameters( + command=sys.executable, + args=[ + str(CORE / "bin/squad"), "mcp", "serve", + "--runtime-dir", str(runtime), "--surface", "codex-app", + ], + cwd=ROOT, + ) + async with Client(parameters) as client: + status = await client.call_tool( + "squad_status", {"run_id": started["run_id"]}, + ) + self.assertFalse(status.is_error) + self.assertEqual( + status.structured_content["data"]["run_id"], started["run_id"], + ) + self.assertIn( + status.structured_content["data"]["state"], {"queued", "running"}, + ) + + asyncio.run(observe_then_disconnect()) + deadline = time.monotonic() + 8 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "succeeded": + break + time.sleep(0.05) + self.assertEqual(service.status(started["run_id"])["state"], "succeeded") + class InstalledWheelMCPBoundaryTest(unittest.TestCase): @staticmethod From 04145d277b81d909b6a749d3f5ae4262c27d32b5 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 06:59:40 +0530 Subject: [PATCH 076/197] test: add M4 cross-surface acceptance probe --- test/core/probes/m4_cross_surface.py | 379 +++++++++++++++++++++++++++ 1 file changed, 379 insertions(+) create mode 100644 test/core/probes/m4_cross_surface.py diff --git a/test/core/probes/m4_cross_surface.py b/test/core/probes/m4_cross_surface.py new file mode 100644 index 0000000..ea977a7 --- /dev/null +++ b/test/core/probes/m4_cross_surface.py @@ -0,0 +1,379 @@ +#!/usr/bin/env python3 +"""Prepare and finish one private M4 cross-surface MCP acceptance run.""" + +from __future__ import annotations + +import argparse +import asyncio +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import time +from typing import Any + +from devsquad.service import Service +from devsquad.store import request_hash + + +TERMINAL_STATES = {"succeeded", "failed", "cancelled", "timed_out"} + + +def _git(repo: Path, *arguments: str) -> str: + return subprocess.run( + ["git", "-C", str(repo), *arguments], + check=True, + text=True, + capture_output=True, + ).stdout.strip() + + +def _write_json(path: Path, value: dict[str, Any]) -> None: + path.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") + path.chmod(0o600) + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(65536), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _wait(service: Service, run_id: str, states: set[str], timeout: int = 20) -> dict[str, Any]: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] in states: + return status + time.sleep(0.05) + raise TimeoutError(f"run did not reach {sorted(states)}: {service.status(run_id)}") + + +def _routing_documents() -> tuple[dict[str, Any], dict[str, Any]]: + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "m4-offline-reviewer", + "harness": "fixture", + "model_family": "offline-fixture", + "model_id": "m4-offline-review", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "m4-offline", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["m4-cross-surface-probe"], + }], + "bindings": { + "review.deep": {"profile_id": "m4-offline-reviewer", "version": 1}, + }, + } + policy = { + "schema_version": 1, + "id": "m4-cross-surface-policy", + "version": 1, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"m4-cross-surface": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "m4-offline": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + return profiles, policy + + +def _make_repository(repo: Path) -> tuple[str, str]: + repo.mkdir() + _git(repo, "init", "-q") + _git(repo, "config", "user.email", "devsquad-probe@example.invalid") + _git(repo, "config", "user.name", "DevSquad Probe") + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "devsquad").mkdir() + (repo / "src/value.py").write_text("VALUE = 'base'\n") + (repo / "tests/test_value.py").write_text("# bounded fixture\n") + profiles, policy = _routing_documents() + _write_json(repo / "devsquad/profiles.json", profiles) + _write_json(repo / "devsquad/policy.json", policy) + _git(repo, "add", ".") + _git(repo, "commit", "-qm", "probe base") + base = _git(repo, "rev-parse", "HEAD") + (repo / "src/value.py").write_text("VALUE = 'candidate'\n") + _git(repo, "add", "src/value.py") + _git(repo, "commit", "-qm", "probe candidate") + return base, _git(repo, "rev-parse", "HEAD") + + +def _task(repo: Path, base: str, target: str) -> dict[str, Any]: + return { + "schema_version": 1, + "project": { + "repo_path": str(repo), "base_ref": base, "target_ref": target, + }, + "workflow": "branch-review", + "goal": "Prove that local MCP clients share one durable saved run.", + "task_class": "m4-cross-surface", + "acceptance": [{ + "id": "shared-run", + "description": "Each client observes the same run and bound evidence.", + "evidence_kind": "review", + }], + "checks": [{ + "id": "offline-check", + "argv": [sys.executable, "-c", "print('m4 check passed')"], + "cwd": ".", + "timeout_seconds": 15, + "required_to_pass": True, + }], + "scope": {"read_paths": ["src", "tests"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": 120, + "max_worker_invocations": 1, + "max_revisions": 0, + "max_fallbacks_per_step": 0, + }, + "review": {"mode": "standard"}, + "origin": {"surface": "terminal-m4-probe"}, + } + + +def prepare(args: argparse.Namespace) -> int: + stamp = time.strftime("%Y%m%dT%H%M%SZ", time.gmtime()) + run_dir = args.output_dir.expanduser().resolve() / f"m4-cross-surface-{stamp}" + run_dir.mkdir(parents=True, mode=0o700, exist_ok=False) + run_dir.chmod(0o700) + repo = run_dir / "repository" + base, target = _make_repository(repo) + task = _task(repo, base, target) + fixture = { + "verdict": "findings", + "summary": "The bounded fixture changes the configured value.", + "findings": [{ + "id": "M4-1", + "severity": "low", + "title": "Fixture value changed", + "description": "The candidate intentionally changes the fixture value.", + "path": "src/value.py", + "start_line": 1, + "end_line": 1, + "evidence": "The candidate contains VALUE = 'candidate'.", + }], + } + service = Service(args.runtime) + started = service.start( + task, + f"m4-cross-surface-{stamp}", + _internal_review_fixture=fixture, + ) + waiting = _wait(service, started["run_id"], {"awaiting_host", "failed"}) + if waiting["state"] != "awaiting_host": + raise RuntimeError(f"offline run did not produce a handoff: {waiting}") + state = { + "schema_version": 1, + "run_dir": str(run_dir), + "runtime": str(args.runtime.resolve()), + "squad_executable": str(args.squad.resolve(strict=True)), + "run_id": started["run_id"], + "terminal_start": started, + "waiting": waiting, + "base_oid": base, + "target_oid": target, + } + state_path = run_dir / "probe-state.json" + _write_json(state_path, state) + print(json.dumps({ + "state_file": str(state_path), + "run_id": started["run_id"], + "state": waiting["state"], + "version": waiting["version"], + }, sort_keys=True)) + return 0 + + +def _data(result: Any) -> dict[str, Any]: + payload = result.structured_content + if not isinstance(payload, dict) or payload.get("ok") is not True: + raise RuntimeError(f"MCP operation failed: {payload}") + return payload["data"] + + +async def _client( + state: dict[str, Any], surface: str, +): + from mcp import Client, StdioServerParameters + + parameters = StdioServerParameters( + command=state["squad_executable"], + args=[ + "mcp", "serve", "--runtime-dir", state["runtime"], + "--surface", surface, + ], + ) + return Client(parameters) + + +async def _complete(state: dict[str, Any]) -> dict[str, Any]: + run_id = state["run_id"] + codex = await _client(state, "codex-app") + async with codex: + codex_status = _data(await codex.call_tool("squad_status", {"run_id": run_id})) + codex_events = _data(await codex.call_tool( + "squad_events", {"run_id": run_id, "after": 0, "limit": 100}, + )) + + claude = await _client(state, "claude-code") + async with claude: + claude_status = _data(await claude.call_tool("squad_status", {"run_id": run_id})) + claude_events = _data(await claude.call_tool( + "squad_events", {"run_id": run_id, "after": 0, "limit": 100}, + )) + claimed = _data(await claude.call_tool("squad_handoff_claim", { + "run_id": run_id, + "expected_version": claude_status["version"], + "owner": "m4-claude-client", + })) + + competing = await _client(state, "antigravity") + async with competing: + fenced_result = await competing.call_tool("squad_handoff_claim", { + "run_id": run_id, + "expected_version": claude_status["version"], + "owner": "m4-competing-client", + }) + fenced = fenced_result.structured_content + + packet = claimed["handoff"]["packet"] + decision_body = { + "schema_version": 1, + "submission_id": "m4-cross-surface-accept", + "disposition": "accept", + "reason": "Accept the bounded offline evidence after cross-client inspection.", + "evidence_refs": [{ + "artifact_id": item["artifact_id"], "sha256": item["sha256"], + } for item in packet["artifacts"]], + } + decision = {**decision_body, "submission_hash": request_hash(decision_body)} + claude_complete = await _client(state, "claude-code") + async with claude_complete: + completed = _data(await claude_complete.call_tool( + "squad_handoff_complete", + {"run_id": run_id, "claim": claimed["claim"], "decision": decision}, + )) + + codex_result = await _client(state, "codex-app") + async with codex_result: + result = _data(await codex_result.call_tool( + "squad_result", {"run_id": run_id, "preview_bytes": 0}, + )) + final_events = _data(await codex_result.call_tool( + "squad_events", {"run_id": run_id, "after": 0, "limit": 100}, + )) + return { + "codex_status": codex_status, + "claude_status": claude_status, + "preclaim_ledgers_identical": codex_events == claude_events, + "claim": claimed["claim"], + "packet_sha256": claimed["handoff"]["packet_sha256"], + "candidate_sha256": packet["candidate_sha256"], + "competing_claim": fenced, + "completed": completed, + "result": result, + "final_events": final_events, + } + + +def complete(args: argparse.Namespace) -> int: + state_path = args.state_file.expanduser().resolve(strict=True) + state = json.loads(state_path.read_text()) + evidence = asyncio.run(_complete(state)) + if not evidence["preclaim_ledgers_identical"]: + raise RuntimeError("Codex and Claude clients observed different ledgers") + fenced = evidence["competing_claim"] + if (not isinstance(fenced, dict) or fenced.get("ok") is not False + or fenced.get("error", {}).get("code") != "CONFLICT"): + raise RuntimeError(f"competing host was not fenced: {fenced}") + if evidence["completed"]["state"] != "succeeded": + raise RuntimeError(f"completion did not succeed: {evidence['completed']}") + result = evidence["result"] + if result["run_id"] != state["run_id"] or not result["ready"]: + raise RuntimeError(f"terminal result is not ready: {result}") + receipt = { + "schema_version": 1, + "status": "passed", + "run_id": state["run_id"], + "terminal_start": state["terminal_start"], + "clients": ["codex-app", "claude-code", "antigravity"], + "same_run_id": ( + evidence["codex_status"]["run_id"] + == evidence["claude_status"]["run_id"] + == result["run_id"] + == state["run_id"] + ), + "preclaim_ledgers_identical": True, + "packet_sha256": evidence["packet_sha256"], + "candidate_sha256": evidence["candidate_sha256"], + "competing_claim_error": fenced["error"]["code"], + "terminal_state": result["state"], + "event_count": len(evidence["final_events"]["events"]), + "artifacts": [{ + key: artifact[key] + for key in ("id", "name", "sha256", "byte_size") + } for artifact in result["artifacts"]], + } + receipt_path = Path(state["run_dir"]) / "probe-receipt.json" + _write_json(receipt_path, receipt) + print(json.dumps({ + "status": "passed", + "run_id": state["run_id"], + "receipt": str(receipt_path), + "receipt_sha256": _sha256(receipt_path), + }, sort_keys=True)) + return 0 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--runtime", type=Path, + default=Path.home() / ".devsquad/runtime", + ) + parser.add_argument( + "--squad", type=Path, + default=Path.home() / ".local/bin/squad", + ) + parser.add_argument( + "--output-dir", type=Path, + default=Path.home() / ".devsquad/private-probes", + ) + subparsers = parser.add_subparsers(dest="operation", required=True) + prepare_parser = subparsers.add_parser("prepare") + prepare_parser.set_defaults(function=prepare) + complete_parser = subparsers.add_parser("complete") + complete_parser.add_argument("--state-file", type=Path, required=True) + complete_parser.set_defaults(function=complete) + args = parser.parse_args() + args.runtime = args.runtime.expanduser().resolve() + args.squad = args.squad.expanduser().resolve(strict=True) + args.output_dir = args.output_dir.expanduser().resolve() + args.output_dir.mkdir(parents=True, mode=0o700, exist_ok=True) + return args.function(args) + + +if __name__ == "__main__": + raise SystemExit(main()) From aa0fef5e6e734f51bb1ba6fbcfa9e659daea6e01 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 07:04:01 +0530 Subject: [PATCH 077/197] fix: annotate MCP tool side effects --- plugin/core/src/devsquad/mcp_server.py | 44 ++++++++++++++++++++------ test/core/test_mcp.py | 15 +++++++++ 2 files changed, 50 insertions(+), 9 deletions(-) diff --git a/plugin/core/src/devsquad/mcp_server.py b/plugin/core/src/devsquad/mcp_server.py index 679d278..6c00cea 100644 --- a/plugin/core/src/devsquad/mcp_server.py +++ b/plugin/core/src/devsquad/mcp_server.py @@ -234,6 +234,32 @@ def build_server( """Build the local stdio server without starting it.""" server_type = _server_type() + from mcp.types import ToolAnnotations + + read_only = ToolAnnotations( + readOnlyHint=True, + destructiveHint=False, + idempotentHint=True, + openWorldHint=False, + ) + durable_write = ToolAnnotations( + readOnlyHint=False, + destructiveHint=False, + idempotentHint=True, + openWorldHint=False, + ) + state_change = ToolAnnotations( + readOnlyHint=False, + destructiveHint=False, + idempotentHint=False, + openWorldHint=False, + ) + destructive_change = ToolAnnotations( + readOnlyHint=False, + destructiveHint=True, + idempotentHint=True, + openWorldHint=False, + ) bridge = MCPBridge( runtime, service, @@ -251,13 +277,13 @@ def build_server( ), ) - @server.tool(name="squad_doctor") + @server.tool(name="squad_doctor", annotations=read_only) def squad_doctor() -> dict[str, Any]: """Inspect versions, capabilities and local host registration drift.""" return bridge.doctor() - @server.tool(name="squad_start") + @server.tool(name="squad_start", annotations=durable_write) def squad_start( task: dict[str, Any], idempotency_key: str, @@ -267,13 +293,13 @@ def squad_start( return bridge.start(task, idempotency_key, supersedes_run_id) - @server.tool(name="squad_status") + @server.tool(name="squad_status", annotations=read_only) def squad_status(run_id: str) -> dict[str, Any]: """Inspect state, version, active work and the next action.""" return bridge.status(run_id) - @server.tool(name="squad_events") + @server.tool(name="squad_events", annotations=read_only) def squad_events( run_id: str, after: int = 0, limit: int = 100, ) -> dict[str, Any]: @@ -281,7 +307,7 @@ def squad_events( return bridge.events(run_id, after, limit) - @server.tool(name="squad_result") + @server.tool(name="squad_result", annotations=read_only) def squad_result( run_id: str, preview_bytes: int = 4096, ) -> dict[str, Any]: @@ -289,13 +315,13 @@ def squad_result( return bridge.result(run_id, preview_bytes) - @server.tool(name="squad_cancel") + @server.tool(name="squad_cancel", annotations=destructive_change) def squad_cancel(run_id: str) -> dict[str, Any]: """Persist cancellation intent for a saved run.""" return bridge.cancel(run_id) - @server.tool(name="squad_resume") + @server.tool(name="squad_resume", annotations=state_change) def squad_resume( run_id: str, recovery: dict[str, Any] | None = None, ) -> dict[str, Any]: @@ -303,7 +329,7 @@ def squad_resume( return bridge.resume(run_id, recovery) - @server.tool(name="squad_handoff_claim") + @server.tool(name="squad_handoff_claim", annotations=state_change) def squad_handoff_claim( run_id: str, expected_version: int, @@ -314,7 +340,7 @@ def squad_handoff_claim( return bridge.handoff_claim(run_id, expected_version, owner, prior_claim) - @server.tool(name="squad_handoff_complete") + @server.tool(name="squad_handoff_complete", annotations=destructive_change) def squad_handoff_complete( run_id: str, claim: dict[str, Any], diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index 82cd849..18f6d32 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -670,6 +670,21 @@ async def probe(): self.assertEqual( tools["squad_status"].input_schema["required"], ["run_id"], ) + status_annotations = tools["squad_status"].annotations.model_dump( + by_alias=True, + ) + self.assertEqual(status_annotations, { + "title": None, + "readOnlyHint": True, + "destructiveHint": False, + "idempotentHint": True, + "openWorldHint": False, + }) + cancel_annotations = tools["squad_cancel"].annotations.model_dump( + by_alias=True, + ) + self.assertTrue(cancel_annotations["destructiveHint"]) + self.assertFalse(cancel_annotations["readOnlyHint"]) status = await client.call_tool("squad_status", {"run_id": "run-1"}) self.assertFalse(status.is_error) self.assertEqual(status.structured_content["data"]["version"], 2) From 06ccf27fd91c839893d8efd73340fa14b1409ad5 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 07:11:51 +0530 Subject: [PATCH 078/197] docs: record M4 cross-surface acceptance --- docs/plans/engineering-team/RESUME.md | 51 ++++---- docs/plans/engineering-team/backlog.json | 15 ++- .../evidence/M4-local-mcp-2026-09-23.json | 111 ++++++++++++++++++ 3 files changed, 150 insertions(+), 27 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/M4-local-mcp-2026-09-23.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index bffacb1..4c8ed50 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -50,18 +50,19 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne worker/delegation guard. The gate is 200 core tests with `ResourceWarning` promoted to failure, 208 Bash assertions, and 12 focused tests against the installed official SDK. -- M4 Plan 06-02 is complete at `9abad1e`. Four packaged host templates now - drive duplicate-aware `squad setup`; `squad doctor` and the ninth MCP tool, - `squad_doctor`, report the app-loaded command, SDK and registration drift - without returning environment maps or arbitrary arguments. Registration is - fail-closed for inherited/duplicate/malformed config and is idempotent while - preserving unrelated settings. The gate is 211 core tests with - `ResourceWarning` promoted to failure, 208 Bash assertions and 21 focused - tests under the pinned SDK. Installed Codex 0.135.0, Claude 2.1.220, - Antigravity 1.2.3 and Grok 0.2.111 CLIs each passed add, second-run no-op and - drift-repair checks under an isolated temporary HOME. That is CLI/config - evidence, not a claim of in-app operation; real account configs remain - untouched. M4 remains in progress for Plan 06-03 cross-surface receipts. +- M4 Plan 06-03 has completed all independent work at `aa0fef5`. The stable + isolated runtime at `~/.devsquad/releases/0.1.0+aa0fef5` is registered in + all four real local host configs; `squad doctor` reports four installed and + four matching registrations, and a second setup pass made no changes. One + terminal-started saved run was inspected through the real Codex MCP host, + then claimed/completed through official stdio SDK clients with identical + ledger identity, a rejected competing claim and terminal receipt hashes. + Closing the MCP client while a detached worker ran did not terminate it. + The gate is 212 core tests, 208 Bash assertions and 22 pinned-SDK tests. + M4 remains **blocked**, not complete, because its exact gate still requires + a normally authenticated Claude Code host to perform the real handoff; the + labeled SDK client is deliberately not presented as that proof. See the + [portable redacted evidence](evidence/M4-local-mcp-2026-09-23.json). - Last-observed provider readiness outside accepted M3: Claude CLI is not logged in; Grok CLI authentication expired; Gemini CLI's individual-account path is unsupported and its supported successor is Antigravity; Antigravity is @@ -75,8 +76,8 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne planning agents all hit the same Plus limit; continue locally until shared agent capacity is restored, then use only bounded leaf reviews. - Full assignment remains **M1–M7 plus C1**, as specified in - [SOL-HANDOFF.md](SOL-HANDOFF.md). M4 is next; M5 may proceed after the frozen - M3 service boundary and can be developed alongside M4 in isolated slices. + [SOL-HANDOFF.md](SOL-HANDOFF.md). M5 is next because it depends on accepted + M3, while the external M4 Claude live gate remains recorded and paused. ## Completed and preserved @@ -88,10 +89,11 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 211 tests passed through M4 Plan 06-02 with warnings promoted to errors | +| Python core discovery | 212 tests passed through M4 Plan 06-03 with warnings promoted to errors | | Bash 3.2 regression suite | 10 test files, 208 assertions passed | -| Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 21 official-SDK focused tests passed | -| M4 local host setup | Four installed host CLIs registered under an isolated HOME, repaired drift and made no second-run changes; duplicate/inherited configs fail closed | +| Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | +| M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | +| M4 cross-surface proof | Real Codex read the terminal-started run through MCP; official SDK clients proved identical ledger, fenced claims, completion and disconnect survival; actual Claude handoff remains blocked on login | | Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 8 | | Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | | Integrated native adapter proof | Passed at `97a10f0`; gpt-5.5/low, read-only, correlated completion and confirmed process-group cleanup | @@ -115,7 +117,8 @@ the earlier apparent nonresponses. The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md), [M2-STATUS.md](M2-STATUS.md) and [M3-STATUS.md](M3-STATUS.md). -[backlog.json](backlog.json) marks all three complete and M4 next. +[backlog.json](backlog.json) marks M1–M3 complete, M4 blocked on its external +Claude live gate, and M5 next. Unauthenticated, unsupported or permission-blocked provider paths are not advertised as verified. @@ -123,12 +126,12 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Execute M4 Plan 06-03: install one stable isolated local MCP runtime, use - the idempotent setup path for the available host surfaces, and prove one - saved run across terminal and Codex with client-disconnect survival, - handoff fencing, duplicate detection and installed-version receipts. -3. Keep Claude/Grok/Antigravity probes paused until their normal login or trust - blockers are resolved. They do not block the independent Codex M3 gate. +2. Execute M5 from a requirement-to-evidence matrix: first add the Claude + headless adapter contract, then isolated one-writer delivery, scope/fencing, + different-model review, exact-candidate checks and bounded disposition. +3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login + or trust blockers are resolved. Their live gates remain open, but M5 may + proceed independently from accepted M3. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 05cade9..0459e6c 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -11,7 +11,7 @@ "execution_brief": "SOL-HANDOFF.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", - "next_milestone": "M4", + "next_milestone": "M5", "milestones": [ { "id": "M1", @@ -173,7 +173,7 @@ "id": "M4", "title": "Shared local MCP access and host handoffs", "depends_on": ["M3"], - "status": "in_progress", + "status": "blocked", "acceptance_section": "M4 — Use that same run from local apps", "evidence": [ { @@ -193,9 +193,18 @@ "artifact": "MCP-LOCAL-ACCESS.md", "recorded_at": "2026-09-23T04:22:26+05:30", "availability": "tracked_tests" + }, + { + "kind": "blocked_acceptance", + "revision": "aa0fef5", + "command_or_action": "212 core tests with ResourceWarning promoted to error, 208 shell assertions, 22 pinned-SDK tests, stable four-host setup, real Codex MCP read and a terminal-to-stdio-client saved-run handoff with disconnect survival and claim fencing", + "outcome": "All independent M4 implementation and cross-client gates pass; Codex live operation is proven, while the exact milestone gate remains blocked because Claude Code requires normal provider login before it can perform the real host handoff", + "artifact": "evidence/M4-local-mcp-2026-09-23.json", + "recorded_at": "2026-09-23T06:59:59+05:30", + "availability": "portable_redacted" } ], - "blocker": null + "blocker": "Normal Claude Code provider login is required before the exact real-host handoff acceptance can run; SDK-labeled client evidence is not substituted for that live proof" }, { "id": "M5", diff --git a/docs/plans/engineering-team/evidence/M4-local-mcp-2026-09-23.json b/docs/plans/engineering-team/evidence/M4-local-mcp-2026-09-23.json new file mode 100644 index 0000000..f5d8fd7 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M4-local-mcp-2026-09-23.json @@ -0,0 +1,111 @@ +{ + "schema_version": 1, + "milestone": "M4", + "status": "blocked", + "implementation_revision": "aa0fef5e6e734f51bb1ba6fbcfa9e659daea6e01", + "recorded_at": "2026-09-23T06:59:59+05:30", + "offline_gate": { + "command": "PYTHONDONTWRITEBYTECODE=1 python3 -W error::ResourceWarning -m unittest discover -s test/core -q", + "result": "212 tests passed; the two optional-SDK tests were skipped in the dependency-free base environment", + "optional_sdk_command": "PYTHONDONTWRITEBYTECODE=1 ~/.devsquad/releases/0.1.0+5b82f53/venv/bin/python -W error::ResourceWarning test/core/test_mcp.py -q", + "optional_sdk_result": "22 tests passed, including official in-memory schema calls and a real stdio-client disconnect while a detached worker survived to success", + "bash_command": "bash test/run.sh", + "bash_result": "10 test files and 208 assertions passed", + "clock_note": "One earlier Bash pass reported portable timeout elapsed 369s while the complete suite wall time was 25s; the isolated test and two subsequent complete Bash gates passed without a code change" + }, + "installed_runtime": { + "launcher": "~/.local/bin/squad", + "immutable_release": "~/.devsquad/releases/0.1.0+aa0fef5/venv/bin/squad", + "devsquad_core": "0.1.0", + "mcp_sdk": "2.2.0", + "pip_check": "passed", + "doctor": "ready; 4 installed local apps and 4 matching registrations" + }, + "installed_hosts": { + "codex": { + "version": "codex-cli 0.153.4", + "executable": "/Applications/ChatGPT.app/Contents/Resources/codex", + "registration": "matching", + "live_operation": "squad_status succeeded" + }, + "claude_code": { + "version": "2.1.220", + "registration": "matching", + "live_operation": "blocked: normal provider login required" + }, + "antigravity": { + "version": "1.2.3", + "registration": "matching", + "live_operation": "blocked: headless scoped trust/permission unresolved" + }, + "grok": { + "version": "0.2.111", + "registration": "matching", + "live_operation": "blocked: provider authentication expired" + } + }, + "registration_evidence": { + "first_setup": "all four hosts updated to the immutable aa0fef5 release and re-inspected matching", + "second_setup": "all four hosts returned unchanged; configuration hashes were unchanged", + "codex_version_drift": "PATH Codex 0.135.0 could not parse the desktop app's ultra reasoning setting; the explicit template now prefers bundled Codex 0.153.4 without changing the global setting" + }, + "cross_surface_run": { + "run_id": "c88c025e-9dc8-4b5a-ac3b-1d1d0c4c9c20", + "terminal_start_state": "awaiting_host", + "terminal_start_version": 14, + "terminal_state": "succeeded", + "terminal_version": 22, + "packet_sha256": "0b3d04320f3cb1404c3364a0b6fafa6d7d2fd2cd51d58ef601bb860b27fd0453", + "candidate_sha256": "4c911e531fda1d2110294f6485755bc2ddff4d9c414799c48c4a1d6414e828f1", + "event_count": 22, + "same_run_id": true, + "preclaim_ledgers_identical": true, + "competing_claim_error": "CONFLICT", + "sdk_client_surfaces": [ + "codex-app", + "claude-code", + "antigravity" + ], + "private_probe_receipt_sha256": "1d0e5aab27ca18934c9b0b8cf02e5141d0269304121a0f79b6c70ff9eff2f05a", + "terminal_report_sha256": { + "artifact-manifest.json": "754e08a762beed0fc6ee37f9c8e69b580fdeddc3a0d3d96ad1cd0e7364fcb7f6", + "events.jsonl": "748b46de74791af1b01735be8c0636ce54d365806684d83b6dcafc38ed5faec9", + "receipt.json": "4e214880b8f5a6248060bbaf2b62dc46d4d05cafda21404d0ad6fd4bcad956ff", + "receipt.md": "db6a6576653f5f2c2ce0a3f680c4d83ca2b884445f04bffb56810e59c29560da", + "result-receipt.json": "4e214880b8f5a6248060bbaf2b62dc46d4d05cafda21404d0ad6fd4bcad956ff" + } + }, + "codex_live_receipt": { + "successful_jsonl_sha256": "81bd8e9e14451afd93ebfc467c4a5dfc9919b1551699cdb1d3cc66bd6454e893", + "thread_id": "01a0cbdf-b599-71b1-81b9-4eb1df5cd92c", + "model": "gpt-5.5", + "effort": "low", + "permission": "read-only sandbox plus one invocation-scoped approval for squad_status only", + "tool": "squad_status", + "tool_result": { + "run_id": "c88c025e-9dc8-4b5a-ac3b-1d1d0c4c9c20", + "state": "awaiting_host", + "version": 14, + "next_action": "claim_handoff" + }, + "native_usage": { + "input_tokens": 56371, + "cached_input_tokens": 9344, + "output_tokens": 296, + "reasoning_output_tokens": 161 + }, + "failed_approval_probe": { + "jsonl_sha256": "e91644f7d61292c58366682fd3a1f427ed3363eeccfc4bd051daf5de87149c86", + "outcome": "Codex loaded DevSquad and selected squad_status, but non-interactive approval policy blocked the unannotated tool before execution", + "native_usage": { + "input_tokens": 56693, + "cached_input_tokens": 38016, + "output_tokens": 165, + "reasoning_output_tokens": 20 + }, + "closure": "aa0fef5 publishes read-only/idempotent/open-world/destructive annotations for all nine tools; the successful live retry additionally used a one-invocation squad_status approval override" + } + }, + "blocking_gate": "M4 cannot be marked complete until a normally authenticated Claude Code app/CLI performs the required real handoff operation against the saved runtime; SDK-labeled client evidence is retained but is not substituted for that live host proof", + "next_independent_work": "Proceed to M5, which depends on accepted M3 rather than completion of the externally blocked M4 live gate" +} From d96e9e4824c039639e36246aae48fa63f59c0a0e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 14:18:07 +0200 Subject: [PATCH 079/197] feat: add bounded Claude headless adapter --- docs/plans/engineering-team/M5-STATUS.md | 25 +++++ docs/plans/engineering-team/backlog.json | 2 +- plugin/core/adapters/claude/adapter.json | 22 ++++ plugin/core/pyproject.toml | 1 + plugin/core/schemas/adapter.schema.json | 10 +- plugin/core/src/devsquad/adapters.py | 54 ++++++++- plugin/lib/adapter.sh | 2 +- plugin/lib/claude-wrapper.sh | 70 ++++++++++++ test/core/test_adapters.py | 136 +++++++++++++++++++++++ test/test_wrapper_contract.sh | 4 +- 10 files changed, 315 insertions(+), 11 deletions(-) create mode 100644 docs/plans/engineering-team/M5-STATUS.md create mode 100644 plugin/core/adapters/claude/adapter.json create mode 100644 plugin/lib/claude-wrapper.sh create mode 100644 test/core/test_adapters.py diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md new file mode 100644 index 0000000..aea8590 --- /dev/null +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -0,0 +1,25 @@ +# M5 implementation status + +M5 is **in progress**. This matrix is derived from the authoritative M5 +requirements before implementation; a row becomes verified only when its +behavioral evidence exists. The milestone remains incomplete until the live +two-harness gate passes. + +| Requirement | Planned evidence | Status | +|---|---|---| +| Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | in progress | +| Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | pending | +| Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | pending | +| Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | router verified; workflow pending | +| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | pending | +| Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | pending | +| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | pending | +| Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | pending | +| Crash recovery | Killing a live implementation supervisor cannot create a duplicate writer on resume | pending | +| Live acceptance | One bounded issue completes across at least two authenticated subscription harnesses with different verified models | pending | + +## Boundary + +M5 implements only the fixed `issue-delivery` sequence. It does not add an +arbitrary DAG, broad autonomous project implementation, automatic integration, +merge, push or publication. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 0459e6c..673f13e 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -210,7 +210,7 @@ "id": "M5", "title": "Bounded implementation with independent review", "depends_on": ["M3"], - "status": "pending", + "status": "in_progress", "acceptance_section": "M5 — Deliver a bounded engineering change", "evidence": [], "blocker": null diff --git a/plugin/core/adapters/claude/adapter.json b/plugin/core/adapters/claude/adapter.json new file mode 100644 index 0000000..f0fa639 --- /dev/null +++ b/plugin/core/adapters/claude/adapter.json @@ -0,0 +1,22 @@ +{ + "schema_version": 1, + "name": "claude", + "transport": "cli_exec", + "binary_candidates": ["claude"], + "model_provider": "anthropic", + "verified_harness_versions": ["2.1.220 (Claude Code)"], + "capabilities": { + "efforts_by_model": {}, + "native_model_list": false, + "resume": true + }, + "permission_profiles": { + "read_only": ["--permission-mode", "plan", "--tools", "Read,Glob,Grep"], + "workspace_write": ["--permission-mode", "acceptEdits", "--tools", "Read,Glob,Grep,Edit,Write"] + }, + "permission_tools": { + "read_only": ["Read", "Glob", "Grep"], + "workspace_write": ["Read", "Glob", "Grep", "Edit", "Write"] + }, + "output_format": "json" +} diff --git a/plugin/core/pyproject.toml b/plugin/core/pyproject.toml index b469d18..9d4222b 100644 --- a/plugin/core/pyproject.toml +++ b/plugin/core/pyproject.toml @@ -27,6 +27,7 @@ where = ["src"] "share/devsquad/adapters/codex" = ["adapters/codex/adapter.json"] "share/devsquad/adapters/antigravity" = ["adapters/antigravity/adapter.json"] "share/devsquad/adapters/grok" = ["adapters/grok/adapter.json"] +"share/devsquad/adapters/claude" = ["adapters/claude/adapter.json"] "share/devsquad/adapters" = ["adapters/classification-policy.conf"] "share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/check-result.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/profiles.schema.json", "schemas/review-result.schema.json", "schemas/task.schema.json"] "share/devsquad/profiles" = ["profiles/templates.json"] diff --git a/plugin/core/schemas/adapter.schema.json b/plugin/core/schemas/adapter.schema.json index b8fb04b..af42568 100644 --- a/plugin/core/schemas/adapter.schema.json +++ b/plugin/core/schemas/adapter.schema.json @@ -10,6 +10,14 @@ "transport": {"enum": ["cli_exec", "native_protocol"]}, "binary_candidates": {"type": "array", "minItems": 1, "items": {"type": "string"}}, "capabilities": {"type": "object"}, - "permission_profiles": {"type": "object"} + "permission_profiles": {"type": "object"}, + "permission_tools": { + "type": "object", + "additionalProperties": { + "type": "array", + "uniqueItems": true, + "items": {"type": "string", "minLength": 1} + } + } } } diff --git a/plugin/core/src/devsquad/adapters.py b/plugin/core/src/devsquad/adapters.py index 52166a7..f87171d 100644 --- a/plugin/core/src/devsquad/adapters.py +++ b/plugin/core/src/devsquad/adapters.py @@ -40,6 +40,7 @@ class AdapterManifest: model_provider: str | None efforts_by_model: dict[str, tuple[str, ...]] permission_profiles: dict[str, tuple[str, ...]] + permission_tools: dict[str, tuple[str, ...]] output_format: str verified_versions: tuple[str, ...] @@ -57,6 +58,7 @@ def load(cls, path: Path) -> "AdapterManifest": model_provider=raw.get("model_provider"), efforts_by_model={k: tuple(v) for k, v in capabilities.get("efforts_by_model", {}).items()}, permission_profiles={k: tuple(v) for k, v in raw.get("permission_profiles", {}).items()}, + permission_tools={k: tuple(v) for k, v in raw.get("permission_tools", {}).items()}, output_format=raw.get("output_format", "text"), verified_versions=tuple(raw.get("verified_harness_versions", [])), ) @@ -65,7 +67,17 @@ def resolve_binary(self) -> str | None: return next((p for name in self.binary_candidates if (p := shutil.which(name))), None) def with_model_efforts(self, mapping: dict[str, tuple[str, ...]]) -> "AdapterManifest": - return AdapterManifest(self.name, self.transport, self.binary_candidates, self.model_provider, mapping, self.permission_profiles, self.output_format, self.verified_versions) + return AdapterManifest( + self.name, + self.transport, + self.binary_candidates, + self.model_provider, + mapping, + self.permission_profiles, + self.permission_tools, + self.output_format, + self.verified_versions, + ) def _permission_args(manifest: AdapterManifest, permission: str) -> tuple[str, ...]: @@ -77,7 +89,8 @@ def _permission_args(manifest: AdapterManifest, permission: str) -> tuple[str, . def prepare_cli( manifest: AdapterManifest, *, prompt: str, cwd: str, model: str | None, - effort: str | None, permission: str, timeout_seconds: int, stdin_path: str | None = None, + effort: str | None, permission: str, timeout_seconds: int, + stdin_path: str | None = None, harness_version_value: str | None = None, ) -> LaunchSpec: binary = manifest.resolve_binary() if not binary: @@ -86,6 +99,14 @@ def prepare_cli( supported = manifest.efforts_by_model.get(model or "") if supported is None or effort not in supported: raise ProfileUnsupported(f"unsupported or unverified effort {effort!r} for {manifest.name} model {model!r}") + verification = "unverified" + if harness_version_value is not None and manifest.verified_versions: + if harness_version_value not in manifest.verified_versions: + raise ProfileUnsupported( + f"unverified {manifest.name} harness version: {harness_version_value}" + ) + verification = "verified" + permission_args = _permission_args(manifest, permission) args: list[str] if manifest.name == "codex": sandbox = "read-only" if permission == "read_only" else "workspace-write" @@ -107,13 +128,34 @@ def prepare_cli( args += ["--model", model] if effort: args += ["--reasoning-effort", effort] + elif manifest.name == "claude": + args = [ + binary, + "--print", + "--output-format", "json", + "--safe-mode", + "--disable-slash-commands", + "--no-session-persistence", + "--strict-mcp-config", + "--mcp-config", '{"mcpServers":{}}', + "--no-chrome", + ] + if model: + args += ["--model", model] + if effort: + args += ["--effort", effort] + args.extend(permission_args) + args += [prompt] + permission_args = () else: raise ContractError(f"no argv builder for adapter: {manifest.name}") - args.extend(_permission_args(manifest, permission)) + args.extend(permission_args) requested = ExecutionIdentity( - harness=manifest.name, harness_version=None, model_provider=manifest.model_provider, + harness=manifest.name, harness_version=harness_version_value, + model_provider=manifest.model_provider, model_family=None, model=model, effort=effort, permissions=permission, - verification="unverified", + tools=manifest.permission_tools.get(permission, ()), + verification=verification, ) return LaunchSpec(SCHEMA_VERSION, manifest.name, "cli_exec", tuple(args), str(Path(cwd).resolve()), stdin_path, timeout_seconds, requested, {"DEVSQUAD_WORKER": "1"}) @@ -185,7 +227,7 @@ def classify_cli(spec: LaunchSpec, *, returncode: int, stdout: str, stderr: str, status, code = "failed", "CLI_ERROR" elif not stdout.strip(): status, code = "malformed", "CLI_ERROR" - elif spec.adapter in {"codex", "antigravity", "grok"}: + elif spec.adapter in {"codex", "antigravity", "grok", "claude"}: try: _, deliverable, terminal, provider_error = _provider_records(spec.adapter, stdout) if provider_error: diff --git a/plugin/lib/adapter.sh b/plugin/lib/adapter.sh index 2a212d2..22df821 100644 --- a/plugin/lib/adapter.sh +++ b/plugin/lib/adapter.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # lib/adapter.sh -- Shared CLI adapter core for DevSquad wrappers (D4). -# Sourced by gemini-/codex-/grok-wrapper.sh. Do not execute directly. +# Sourced by gemini-/codex-/grok-/claude-wrapper.sh. Do not execute directly. # # Contract (enforced by test/test_wrapper_contract.sh): # success: response on stdout, exit 0 diff --git a/plugin/lib/claude-wrapper.sh b/plugin/lib/claude-wrapper.sh new file mode 100644 index 0000000..2ef632d --- /dev/null +++ b/plugin/lib/claude-wrapper.sh @@ -0,0 +1,70 @@ +#!/usr/bin/env bash +# lib/claude-wrapper.sh -- Claude Code headless adapter configuration. +# Sourced by agent system prompts. Do not execute directly. +# Shared invocation core lives in lib/adapter.sh (D4 contract). +set -euo pipefail + +_CLAUDE_LIB_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +# shellcheck source=/dev/null +source "${_CLAUDE_LIB_DIR}/adapter.sh" + +_resolve_claude_model() { + _adapter_resolve_model "claude_model" +} + +_claude_configure_adapter() { + ADAPTER_AGENT="claude" + ADAPTER_PREF_MODEL_KEY="claude_model" + ADAPTER_AUTH_HINT="Run 'claude auth login' through the normal subscription flow, then retry." + ADAPTER_FALLBACK="Fallback: use a separately qualified DevSquad profile." + ADAPTER_MISSING_MSG="Claude Code CLI not installed. Install it through the official Claude Code setup." + ADAPTER_EXTRA_AUTH_RE='not logged in|please run /login|login required' + ADAPTER_STDIN_FILE="" + ADAPTER_EXTRA_CHARS_IN=0 + + _adapter_resolve_cli() { + if command -v claude &>/dev/null; then + echo "claude" + else + echo "" + fi + } + + _adapter_build_args() { + ADAPTER_ARGS=( + "--print" + "--output-format" "text" + "--safe-mode" + "--disable-slash-commands" + "--no-session-persistence" + "--strict-mcp-config" + "--mcp-config" '{"mcpServers":{}}' + "--no-chrome" + "--permission-mode" "plan" + "--tools" "Read,Glob,Grep" + ) + if [[ -n "$2" ]]; then + ADAPTER_ARGS+=("--model" "$2") + fi + ADAPTER_ARGS+=("$1") + } +} + +# Usage: invoke_claude "prompt" [word_limit] [timeout_secs] +# This compatibility entry point is deliberately read-only. The durable M5 +# implementer uses the manifest-driven workspace_write profile instead. +invoke_claude() { + local prompt="$1" + local word_limit="${2:-300}" + local timeout_secs="${3:-180}" + local final_prompt="$prompt" + + if [[ "$word_limit" -gt 0 ]] 2>/dev/null; then + if ! echo "$prompt" | grep -qiE 'under [0-9]+ (words|lines)|[0-9]+ (words|lines) max'; then + final_prompt="${prompt}. Under ${word_limit} words." + fi + fi + + _claude_configure_adapter + _adapter_invoke "$final_prompt" "$timeout_secs" +} diff --git a/test/core/test_adapters.py b/test/core/test_adapters.py new file mode 100644 index 0000000..0a0cda6 --- /dev/null +++ b/test/core/test_adapters.py @@ -0,0 +1,136 @@ +from __future__ import annotations + +import json +import os +import stat +import sys +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.adapters import AdapterManifest, classify_cli, prepare_cli +from devsquad.contracts import ProfileUnsupported + + +class ClaudeAdapterTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.binary = Path(self.temp.name) / "claude" + self.binary.write_text("#!/bin/sh\nexit 0\n") + self.binary.chmod(self.binary.stat().st_mode | stat.S_IXUSR) + self.manifest = AdapterManifest.load( + CORE / "adapters" / "claude" / "adapter.json" + ) + + def prepare( + self, + *, + permission: str = "read_only", + model: str | None = None, + effort: str | None = None, + version: str | None = None, + ): + manifest = self.manifest + if effort is not None: + manifest = manifest.with_model_efforts({model: (effort,)}) + with patch.dict(os.environ, {"PATH": str(self.binary.parent)}): + return prepare_cli( + manifest, + prompt="Fix src/My Parser.ts without delegating.", + cwd=self.temp.name, + model=model, + effort=effort, + permission=permission, + timeout_seconds=17, + harness_version_value=version, + ) + + def test_read_only_argv_is_structured_and_has_no_bypass_or_delegation(self): + spec = self.prepare() + self.assertEqual(spec.adapter, "claude") + self.assertEqual(spec.transport, "cli_exec") + self.assertEqual(spec.requested.permissions, "read_only") + self.assertEqual(spec.requested.tools, ("Read", "Glob", "Grep")) + self.assertIn("plan", spec.argv) + self.assertIn("Read,Glob,Grep", spec.argv) + self.assertIn('{"mcpServers":{}}', spec.argv) + self.assertIn("Fix src/My Parser.ts without delegating.", spec.argv) + self.assertNotIn("--dangerously-skip-permissions", spec.argv) + self.assertNotIn("Agent", ",".join(spec.argv)) + self.assertNotIn("Bash", ",".join(spec.argv)) + + def test_workspace_write_argv_and_version_are_exactly_verified(self): + spec = self.prepare( + permission="workspace_write", + model="claude-fixture-1", + effort="high", + version="2.1.220 (Claude Code)", + ) + self.assertIn("acceptEdits", spec.argv) + self.assertIn("Read,Glob,Grep,Edit,Write", spec.argv) + self.assertEqual(spec.requested.model, "claude-fixture-1") + self.assertEqual(spec.requested.effort, "high") + self.assertEqual(spec.requested.harness_version, "2.1.220 (Claude Code)") + self.assertEqual(spec.requested.verification, "verified") + with self.assertRaisesRegex(ProfileUnsupported, "unverified claude"): + self.prepare(version="2.2.0 (Claude Code)") + + def test_structured_result_and_faults_are_not_conflated_with_acceptance(self): + spec = self.prepare() + success = classify_cli( + spec, + returncode=0, + stdout=json.dumps({ + "type": "result", + "subtype": "success", + "is_error": False, + "result": "bounded implementation summary", + "session_id": "fixture-session", + }), + stderr="", + ) + self.assertEqual(success.execution_status, "succeeded") + self.assertEqual(success.acceptance_status, "not_evaluated") + + auth = classify_cli( + spec, + returncode=0, + stdout=json.dumps({ + "type": "result", + "is_error": True, + "result": "401 quota authorization required", + }), + stderr="", + ) + self.assertEqual(auth.execution_status, "failed") + self.assertEqual(auth.error_code, "AUTH_ERROR") + + denied = classify_cli( + spec, + returncode=0, + stdout=json.dumps({ + "type": "result", + "is_error": True, + "result": "tool use denied", + }), + stderr="", + ) + self.assertEqual(denied.execution_status, "denied") + self.assertEqual(denied.error_code, "CLI_ERROR") + + malformed = classify_cli( + spec, + returncode=0, + stdout=json.dumps({"type": "system", "subtype": "init"}), + stderr="", + ) + self.assertEqual(malformed.execution_status, "malformed") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/test_wrapper_contract.sh b/test/test_wrapper_contract.sh index ce591e8..71bc37f 100644 --- a/test/test_wrapper_contract.sh +++ b/test/test_wrapper_contract.sh @@ -18,7 +18,7 @@ ok() { PASS=$((PASS + 1)); } bad() { FAIL=$((FAIL + 1)); echo " FAIL: $1"; } FAKE=$(mktemp -d) -for bin in agy codex grok; do +for bin in agy claude codex grok; do cat > "$FAKE/$bin" <<'FAKESH' #!/bin/bash case "${FAKE_MODE:-success}" in @@ -46,7 +46,7 @@ run_case() { ERR_TXT=$(cat "$errf" 2>/dev/null) } -for spec in "gemini-wrapper.sh:invoke_gemini:gemini" "codex-wrapper.sh:invoke_codex:codex" "grok-wrapper.sh:invoke_grok:grok"; do +for spec in "gemini-wrapper.sh:invoke_gemini:gemini" "claude-wrapper.sh:invoke_claude:claude" "codex-wrapper.sh:invoke_codex:codex" "grok-wrapper.sh:invoke_grok:grok"; do wrapper="${spec%%:*}"; rest="${spec#*:}"; fn="${rest%%:*}"; agent="${rest#*:}" # 1. success: stdout + exit 0 + usage record + contract log From 0d131e8cee3a88d0fa9c33711f360a83dd348bb8 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 14:19:26 +0200 Subject: [PATCH 080/197] docs: checkpoint Claude adapter slice --- docs/plans/engineering-team/M5-STATUS.md | 14 +++++++++++++- docs/plans/engineering-team/RESUME.md | 17 ++++++++++++----- docs/plans/engineering-team/backlog.json | 12 +++++++++++- 3 files changed, 36 insertions(+), 7 deletions(-) diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index aea8590..4322c4a 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -7,7 +7,7 @@ two-harness gate passes. | Requirement | Planned evidence | Status | |---|---|---| -| Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | in progress | +| Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | verified offline at `d96e9e4` | | Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | pending | | Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | pending | | Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | router verified; workflow pending | @@ -23,3 +23,15 @@ two-harness gate passes. M5 implements only the fixed `issue-delivery` sequence. It does not add an arbitrary DAG, broad autonomous project implementation, automatic integration, merge, push or publication. + +## Plan 07-01 checkpoint 1 + +The Claude CLI adapter is available through both the core manifest boundary and +the Bash 3.2 compatibility API. Its verified 2.1.220 profile uses structured +non-interactive output, safe mode, no session persistence, an empty strict MCP +configuration, no browser, explicit role tools and no blanket permission +bypass. The read-only profile exposes only Read/Glob/Grep; the write profile +adds Edit/Write but not Bash or Agent. Offline evidence is 3 focused tests, 215 +full core tests (2 optional-SDK skips), a fresh wheel containing the manifest, +and 220 Bash assertions. This does not claim a live Claude model invocation; +the installed CLI still requires normal provider login. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 4c8ed50..5c2906b 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -78,6 +78,13 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne - Full assignment remains **M1–M7 plus C1**, as specified in [SOL-HANDOFF.md](SOL-HANDOFF.md). M5 is next because it depends on accepted M3, while the external M4 Claude live gate remains recorded and paused. +- M5 Plan 07-01 has started at `d96e9e4`. The core and Bash compatibility + boundaries now include a Claude 2.1.220 headless adapter with structured + output, version-scoped model/effort preparation, explicit Read/Glob/Grep or + Edit/Write tool sets, strict empty MCP configuration, safe mode and no + blanket permission bypass. Its offline gate is 215 core tests, a fresh-wheel + content check and 220 Bash assertions. This is adapter conformance, not a + live Claude model receipt; normal Claude login remains required. ## Completed and preserved @@ -89,8 +96,8 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 212 tests passed through M4 Plan 06-03 with warnings promoted to errors | -| Bash 3.2 regression suite | 10 test files, 208 assertions passed | +| Python core discovery | 215 tests passed through M5 Plan 07-01 adapter slice with warnings promoted to errors | +| Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | | M4 cross-surface proof | Real Codex read the terminal-started run through MCP; official SDK clients proved identical ledger, fenced claims, completion and disconnect survival; actual Claude handoff remains blocked on login | @@ -126,9 +133,9 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Execute M5 from a requirement-to-evidence matrix: first add the Claude - headless adapter contract, then isolated one-writer delivery, scope/fencing, - different-model review, exact-candidate checks and bounded disposition. +2. Continue M5 Plan 07-01 from the committed adapter boundary: add the isolated + one-writer delivery worktree, scope validation and replay-safe local + candidate commit/patch artifacts without merge, push or publication. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 673f13e..238d322 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -212,7 +212,17 @@ "depends_on": ["M3"], "status": "in_progress", "acceptance_section": "M5 — Deliver a bounded engineering change", - "evidence": [], + "evidence": [ + { + "kind": "implementation_checkpoint", + "revision": "d96e9e4", + "command_or_action": "215 core tests with ResourceWarning promoted to error, 220 Bash assertions, 3 focused Claude adapter tests and a fresh wheel content check", + "outcome": "The bounded Claude 2.1.220 headless adapter now uses explicit read/write tool profiles, structured non-interactive output, strict empty MCP config, no blanket permission bypass and shared legacy error classification; delivery workspaces and workflow execution remain open", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-23T12:19:00+05:30", + "availability": "tracked_tests" + } + ], "blocker": null }, { From 6a7e849d957fbd79b23e6f5bad72a6e8f029b82f Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 14:30:06 +0200 Subject: [PATCH 081/197] feat: freeze isolated delivery candidates --- plugin/core/src/devsquad/workspaces.py | 216 ++++++++++++++++++++++++- test/core/test_delivery_workflow.py | 183 +++++++++++++++++++++ 2 files changed, 398 insertions(+), 1 deletion(-) create mode 100644 test/core/test_delivery_workflow.py diff --git a/plugin/core/src/devsquad/workspaces.py b/plugin/core/src/devsquad/workspaces.py index c94de21..0caac8c 100644 --- a/plugin/core/src/devsquad/workspaces.py +++ b/plugin/core/src/devsquad/workspaces.py @@ -12,12 +12,20 @@ from .store import canonical_json, git_common_dir -def _git(repo: Path, *args: str) -> bytes: +MAX_CANDIDATE_PATCH_BYTES = 16 * 1024 * 1024 + + +def _git( + repo: Path, + *args: str, + environment: dict[str, str] | None = None, +) -> bytes: try: result = subprocess.run( ["git", "-C", str(repo), *args], stdout=subprocess.PIPE, stderr=subprocess.PIPE, + env=(None if environment is None else {**os.environ, **environment}), check=False, ) except OSError as exc: @@ -315,3 +323,209 @@ def prepare_check_workspace( "target_oid": target_oid, "scope": list(scopes), } + + +def prepare_delivery_workspace( + source_repo: Path, + runtime: Path, + project_id: str, + run_id: str, + target_oid: str, + read_paths: Iterable[str], + write_paths: Iterable[str], + *, + required_clean_paths: Iterable[str] = (), +) -> dict[str, object]: + """Create or validate one detached, run-owned implementation worktree.""" + repo = source_repo.resolve(strict=True) + reads = tuple(_normalized_relative(path, "read scope path") for path in read_paths) + writes = tuple( + _normalized_relative(path, "write scope path") for path in write_paths + ) + if not writes: + raise ContractError("delivery workspace requires a non-empty write scope") + scopes = tuple(dict.fromkeys((*reads, *writes))) + assert_clean_inputs(repo, scopes, required_clean_paths) + workspace, _ = _prepare_detached_workspace( + repo, runtime, project_id, run_id, target_oid, scopes, + "delivery-worktree", + ) + return { + "schema_version": 1, + "path": str(workspace), + "baseline_oid": target_oid, + "read_scope": list(reads), + "write_scope": list(writes), + } + + +def _assert_delivery_path_scope( + workspace: Path, + changed_paths: Iterable[str], + write_paths: Iterable[str], +) -> tuple[str, ...]: + writes = tuple( + _normalized_relative(path, "write scope path") for path in write_paths + ) + changed = tuple(sorted(dict.fromkeys(changed_paths))) + outside = [ + path for path in changed + if not any(_intersects(path, scope) for scope in writes) + ] + if outside: + raise ContractError( + "delivery candidate changes paths outside write scope: " + + ", ".join(outside) + ) + root = workspace.resolve(strict=True) + for relative in changed: + normalized = _normalized_relative(relative, "candidate path") + candidate = root / normalized + probe = candidate if candidate.exists() or candidate.is_symlink() else candidate.parent + try: + resolved = probe.resolve(strict=False) + except OSError as exc: + raise ContractError( + f"delivery candidate path cannot be resolved: {normalized}" + ) from exc + if resolved != root and root not in resolved.parents: + raise ContractError( + f"delivery candidate path escapes its workspace: {normalized}" + ) + return changed + + +def _candidate_snapshot( + workspace: Path, + baseline_oid: str, + commit_oid: str, + write_paths: Iterable[str], +) -> tuple[dict[str, object], bytes]: + changed = _decode_paths( + _git( + workspace, + "diff", "--no-renames", "--name-only", "-z", + baseline_oid, commit_oid, "--", + ), + "delivery candidate diff", + ) + changed_paths = _assert_delivery_path_scope(workspace, changed, write_paths) + if not changed_paths: + raise ContractError("delivery candidate contains no changes") + captured_untracked = _decode_paths( + _git( + workspace, + "diff", "--no-renames", "--diff-filter=A", "--name-only", "-z", + baseline_oid, commit_oid, "--", + ), + "delivery added-path inventory", + ) + patch = _git( + workspace, + "diff", "--binary", "--no-ext-diff", baseline_oid, commit_oid, "--", + ) + if len(patch) > MAX_CANDIDATE_PATCH_BYTES: + raise ContractError("delivery candidate patch exceeds its byte limit") + tree_oid = _git( + workspace, "rev-parse", "--verify", f"{commit_oid}^{{tree}}", + ).decode().strip() + if (len(tree_oid) != 40 + or any(character not in "0123456789abcdef" for character in tree_oid)): + raise ContractError("delivery candidate tree did not resolve to a full OID") + identity = { + "schema_version": 1, + "baseline_oid": baseline_oid, + "commit_oid": commit_oid, + "tree_oid": tree_oid, + "patch_sha256": hashlib.sha256(patch).hexdigest(), + "changed_paths": list(changed_paths), + } + return { + **identity, + "candidate_sha256": hashlib.sha256( + canonical_json(identity).encode() + ).hexdigest(), + "patch_bytes": len(patch), + "captured_untracked_paths": sorted(captured_untracked), + }, patch + + +def freeze_delivery_candidate( + source_repo: Path, + workspace: Path, + baseline_oid: str, + write_paths: Iterable[str], + run_id: str, +) -> tuple[dict[str, object], bytes]: + """Commit one scoped candidate locally and return its stable patch identity.""" + repo = source_repo.resolve(strict=True) + delivery = workspace.resolve(strict=True) + run = _validate_segment(run_id, "run id") + if delivery.name != "delivery-worktree": + raise ContractError("delivery workspace is not a run-owned delivery worktree") + head_oid = resolve_commit(delivery, "HEAD") + _validate_workspace( + repo, + delivery, + head_oid, + write_paths, + require_clean=False, + ) + dirties = dirty_paths(delivery) + if head_oid != baseline_oid: + if dirties: + raise ContractError("frozen delivery candidate has later workspace changes") + parents = _git( + delivery, "rev-list", "--parents", "-n", "1", head_oid, + ).decode().strip().split() + if parents != [head_oid, baseline_oid]: + raise ContractError("delivery candidate is not a single local baseline commit") + marker = _git( + delivery, "show", "-s", "--format=%s%x00%ae", head_oid, + ).decode("utf-8", "strict").rstrip("\n").split("\0") + if marker != [f"DevSquad candidate {run}", "candidate@devsquad.local"]: + raise ContractError("delivery candidate commit is not coordinator-owned") + return _candidate_snapshot( + delivery, baseline_oid, head_oid, write_paths, + ) + + changed_paths = _assert_delivery_path_scope(delivery, dirties, write_paths) + if not changed_paths: + raise ContractError("delivery candidate contains no changes") + _git(delivery, "add", "-A", "--", ".") + staged = _decode_paths( + _git( + delivery, "diff", "--cached", "--no-renames", "--name-only", "-z", + "--", + ), + "staged delivery candidate", + ) + _assert_delivery_path_scope(delivery, staged, write_paths) + staged_patch = _git( + delivery, "diff", "--cached", "--binary", "--no-ext-diff", "--", + ) + if len(staged_patch) > MAX_CANDIDATE_PATCH_BYTES: + raise ContractError("delivery candidate patch exceeds its byte limit") + commit_environment = { + "GIT_AUTHOR_NAME": "DevSquad Candidate", + "GIT_AUTHOR_EMAIL": "candidate@devsquad.local", + "GIT_COMMITTER_NAME": "DevSquad Candidate", + "GIT_COMMITTER_EMAIL": "candidate@devsquad.local", + } + _git( + delivery, + "-c", "core.hooksPath=/dev/null", + "-c", "commit.gpgSign=false", + "commit", "--quiet", "--no-verify", "--no-gpg-sign", + "-m", f"DevSquad candidate {run}", + environment=commit_environment, + ) + commit_oid = resolve_commit(delivery, "HEAD") + snapshot, patch = _candidate_snapshot( + delivery, baseline_oid, commit_oid, write_paths, + ) + if patch != staged_patch: + raise ContractError("committed delivery patch differs from the staged candidate") + if dirty_paths(delivery): + raise ContractError("delivery workspace remained dirty after candidate commit") + return snapshot, patch diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py new file mode 100644 index 0000000..4b48bbd --- /dev/null +++ b/test/core/test_delivery_workflow.py @@ -0,0 +1,183 @@ +from __future__ import annotations + +import hashlib +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.contracts import ContractError +from devsquad.workspaces import ( + freeze_delivery_candidate, + prepare_delivery_workspace, + resolve_commit, +) + + +class DeliveryWorkspaceTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + self.remote = self.root / "remote.git" + self.repo.mkdir() + self.git(self.repo, "init", "-q") + self.git(self.repo, "config", "user.name", "Fixture") + self.git(self.repo, "config", "user.email", "fixture@example.test") + (self.repo / "src").mkdir() + (self.repo / "tests").mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# base test\n") + (self.repo / "README.md").write_text("fixture\n") + self.git(self.repo, "add", ".") + self.git(self.repo, "commit", "-qm", "base") + self.baseline = resolve_commit(self.repo, "HEAD") + subprocess.run( + ["git", "init", "--bare", "-q", str(self.remote)], check=True, + ) + self.git(self.repo, "remote", "add", "origin", str(self.remote)) + self.git(self.repo, "push", "-q", "origin", "HEAD:refs/heads/main") + self.source_status = self.git(self.repo, "status", "--porcelain") + self.source_refs = self.git(self.repo, "show-ref") + self.remote_refs = self.git(self.remote, "show-ref") + + @staticmethod + def git(repo: Path, *args: str) -> str: + return subprocess.run( + ["git", "-C", str(repo), *args], + check=True, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + ).stdout + + def prepare(self) -> dict[str, object]: + return prepare_delivery_workspace( + self.repo, + self.runtime, + "project-1", + "run-1", + self.baseline, + ("src", "tests"), + ("src/app.py", "tests"), + ) + + def assert_source_unchanged(self) -> None: + self.assertEqual(resolve_commit(self.repo, "HEAD"), self.baseline) + self.assertEqual(self.git(self.repo, "status", "--porcelain"), self.source_status) + self.assertEqual(self.git(self.repo, "show-ref"), self.source_refs) + self.assertEqual(self.git(self.remote, "show-ref"), self.remote_refs) + self.assertEqual((self.repo / "src/app.py").read_text(), "VALUE = 'base'\n") + + def test_scoped_candidate_commit_patch_and_replay_preserve_source_and_remote(self): + prepared = self.prepare() + workspace = Path(prepared["path"]) + self.assertEqual(self.git(workspace, "rev-parse", "--abbrev-ref", "HEAD").strip(), "HEAD") + (workspace / "src/app.py").write_text("VALUE = 'fixed'\n") + (workspace / "tests/test regression.py").write_text( + "def test_regression():\n assert True\n" + ) + + candidate, patch = freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assertEqual(candidate["baseline_oid"], self.baseline) + self.assertEqual(candidate["patch_sha256"], hashlib.sha256(patch).hexdigest()) + self.assertEqual( + candidate["captured_untracked_paths"], ["tests/test regression.py"], + ) + self.assertEqual( + candidate["changed_paths"], ["src/app.py", "tests/test regression.py"], + ) + self.assertIn(b"VALUE = 'fixed'", patch) + self.assertEqual( + self.git(workspace, "rev-parse", "HEAD^").strip(), self.baseline, + ) + self.assertEqual(self.git(workspace, "status", "--porcelain"), "") + replayed, replay_patch = freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assertEqual(replayed, candidate) + self.assertEqual(replay_patch, patch) + self.assert_source_unchanged() + + def test_later_mutation_cannot_replay_a_frozen_candidate(self): + workspace = Path(self.prepare()["path"]) + (workspace / "src/app.py").write_text("VALUE = 'candidate'\n") + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + (workspace / "src/app.py").write_text("VALUE = 'stale'\n") + with self.assertRaisesRegex(ContractError, "later workspace changes"): + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assert_source_unchanged() + + def test_out_of_scope_change_is_rejected_before_commit(self): + workspace = Path(self.prepare()["path"]) + (workspace / "README.md").write_text("unauthorized\n") + with self.assertRaisesRegex(ContractError, "outside write scope"): + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assertEqual(resolve_commit(workspace, "HEAD"), self.baseline) + self.assert_source_unchanged() + + def test_new_symlink_cannot_escape_the_delivery_workspace(self): + workspace = Path(self.prepare()["path"]) + outside = self.root / "outside-secret" + outside.write_text("private\n") + (workspace / "tests/leak").symlink_to(outside) + with self.assertRaisesRegex(ContractError, "escapes its workspace"): + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assertEqual(resolve_commit(workspace, "HEAD"), self.baseline) + self.assert_source_unchanged() + + def test_candidate_requires_a_change(self): + workspace = Path(self.prepare()["path"]) + with self.assertRaisesRegex(ContractError, "contains no changes"): + freeze_delivery_candidate( + self.repo, + workspace, + self.baseline, + ("src/app.py", "tests"), + "run-1", + ) + self.assert_source_unchanged() + + +if __name__ == "__main__": + unittest.main() From f5032d7cbca1df326d3e27c880b933b808df085a Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 14:31:43 +0200 Subject: [PATCH 082/197] docs: checkpoint delivery candidate slice --- docs/plans/engineering-team/M5-STATUS.md | 21 +++++++++++++++++++-- docs/plans/engineering-team/RESUME.md | 17 +++++++++++++---- docs/plans/engineering-team/backlog.json | 9 +++++++++ 3 files changed, 41 insertions(+), 6 deletions(-) diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index 4322c4a..84fdc09 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -8,8 +8,8 @@ two-harness gate passes. | Requirement | Planned evidence | Status | |---|---|---| | Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | verified offline at `d96e9e4` | -| Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | pending | -| Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | pending | +| Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | workspace isolation verified offline at `6a7e849`; workflow writer pending | +| Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `6a7e849` | | Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | router verified; workflow pending | | Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | pending | | Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | pending | @@ -35,3 +35,20 @@ adds Edit/Write but not Bash or Agent. Offline evidence is 3 focused tests, 215 full core tests (2 optional-SDK skips), a fresh wheel containing the manifest, and 220 Bash assertions. This does not claim a live Claude model invocation; the installed CLI still requires normal provider login. + +## Plan 07-01 checkpoint 2 + +The delivery workspace starts from the frozen target in a detached, run-owned +Git worktree. Candidate freezing rejects out-of-scope edits and escaping +symlinks, stages only after validation, disables repository hooks/signing, +creates one coordinator-owned local commit, and returns stable commit/tree/ +patch/candidate hashes plus added-file evidence. A replay returns the same +candidate; any later workspace edit invalidates it. Five focused tests prove +file names with spaces, no-change rejection, scope and symlink failures, +source checkout/index/HEAD preservation and unchanged local-remote refs. + +The complete gate is 220 core tests (2 optional-SDK skips) and 220 Bash +assertions. The first full discovery observed the pre-existing coordinator- +crash race test return `live` once; that isolated test and the full discovery +rerun passed unchanged. Connecting this workspace to the durable implementer +attempt and its existing store-level writer fence remains the next slice. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 5c2906b..83d3f42 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -85,6 +85,15 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne blanket permission bypass. Its offline gate is 215 core tests, a fresh-wheel content check and 220 Bash assertions. This is adapter conformance, not a live Claude model receipt; normal Claude login remains required. +- M5 Plan 07-01 delivery workspace/candidate freezing is complete at `6a7e849`. + A detached run-owned implementation worktree now enforces declared write + scope and symlink containment, creates one coordinator-owned local commit, + returns stable commit/tree/patch/candidate hashes, replays idempotently and + rejects later mutation. Five focused tests prove the original checkout, + index, HEAD and remote refs remain unchanged. The complete gate is 220 core + tests and 220 Bash assertions. One first core run observed the pre-existing + coordinator-crash test return `live`; its isolated run and the complete rerun + passed unchanged. ## Completed and preserved @@ -96,7 +105,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 215 tests passed through M5 Plan 07-01 adapter slice with warnings promoted to errors | +| Python core discovery | 220 tests passed through M5 Plan 07-01 delivery candidate slice with warnings promoted to errors | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -133,9 +142,9 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M5 Plan 07-01 from the committed adapter boundary: add the isolated - one-writer delivery worktree, scope validation and replay-safe local - candidate commit/patch artifacts without merge, push or publication. +2. Continue M5 Plan 07-01 by connecting the isolated delivery worktree to a + durable fenced implementer attempt, then publish its local candidate/patch + artifacts to the saved run without merge, push or publication. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 238d322..0e9fe93 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -221,6 +221,15 @@ "artifact": "M5-STATUS.md", "recorded_at": "2026-09-23T12:19:00+05:30", "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "6a7e849", + "command_or_action": "220 core tests with ResourceWarning promoted to error, 220 Bash assertions and 5 focused delivery-workspace tests", + "outcome": "A run-owned detached delivery worktree now freezes only validated in-scope edits into a replay-safe local candidate commit and binary patch identity while preserving the source checkout and remote refs; durable implementer execution remains open", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-23T18:00:27+05:30", + "availability": "tracked_tests" } ], "blocker": null From 0e88d7343fb0769d711eaf8f5ba5159ead74505f Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 14:53:55 +0200 Subject: [PATCH 083/197] feat: persist fenced delivery candidates --- plugin/core/src/devsquad/delivery_worker.py | 77 ++++++ plugin/core/src/devsquad/detached.py | 56 +++-- plugin/core/src/devsquad/service.py | 94 +++++-- plugin/core/src/devsquad/store.py | 101 ++++++++ plugin/core/src/devsquad/supervisor.py | 150 +++++++++++- plugin/core/src/devsquad/workflows.py | 138 +++++++++++ plugin/core/src/devsquad/workspaces.py | 34 ++- test/core/test_delivery_workflow.py | 256 ++++++++++++++++++++ test/core/test_supervisor.py | 14 ++ 9 files changed, 876 insertions(+), 44 deletions(-) create mode 100644 plugin/core/src/devsquad/delivery_worker.py diff --git a/plugin/core/src/devsquad/delivery_worker.py b/plugin/core/src/devsquad/delivery_worker.py new file mode 100644 index 0000000..ed7e2b6 --- /dev/null +++ b/plugin/core/src/devsquad/delivery_worker.py @@ -0,0 +1,77 @@ +"""Offline bounded implementation worker used for delivery fault injection.""" + +from __future__ import annotations + +import json +from pathlib import Path, PurePosixPath +import sys +import time +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .workflows import make_implementation_evidence + + +MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 +MAX_FIXTURE_WRITES = 100 +MAX_FIXTURE_CONTENT_BYTES = 1024 * 1024 + + +def _relative_path(value: Any) -> str: + if not isinstance(value, str) or not value or "\\" in value or "\0" in value: + raise ContractError("implementation fixture path is invalid") + path = PurePosixPath(value) + if path.is_absolute() or ".." in path.parts or path.as_posix() != value: + raise ContractError("implementation fixture path must be repository-relative") + return value + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("delivery snapshot must be an object") + fixture = snapshot.get("internal_implementation_fixture") + if not isinstance(fixture, dict) or set(fixture) != {"writes", "delay_seconds"}: + raise ContractError("offline implementation fixture is incomplete") + writes, delay = fixture["writes"], fixture["delay_seconds"] + if (not isinstance(writes, list) or not writes + or len(writes) > MAX_FIXTURE_WRITES): + raise ContractError("implementation fixture writes must be a bounded array") + if not isinstance(delay, (int, float)) or isinstance(delay, bool) or not 0 <= delay <= 60: + raise ContractError("implementation fixture delay is invalid") + workspace = Path(snapshot["delivery_workspace"]["path"]).resolve(strict=True) + if delay: + time.sleep(delay) + for item in writes: + if not isinstance(item, dict) or set(item) != {"path", "content"}: + raise ContractError("implementation fixture write fields are invalid") + relative = _relative_path(item["path"]) + content = item["content"] + if (not isinstance(content, str) + or len(content.encode()) > MAX_FIXTURE_CONTENT_BYTES): + raise ContractError("implementation fixture content is invalid") + destination = workspace / relative + resolved = destination.resolve(strict=False) + if resolved != workspace and workspace not in resolved.parents: + raise ContractError("implementation fixture path escapes its workspace") + destination.parent.mkdir(parents=True, exist_ok=True) + destination.write_text(content) + return make_implementation_evidence( + snapshot, "Applied the bounded offline implementation fixture.", + ) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_SNAPSHOT_BYTES + 1) + if len(payload) > MAX_SNAPSHOT_BYTES: + raise ContractError("delivery snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("delivery snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index 42792ad..71a09a2 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -85,16 +85,23 @@ def main(argv=None): stdin_path = None adapter = None handoff = store.handoff_snapshot(args.run_id) + workflow = snapshot["task"]["workflow"] headless_lead = ( snapshot["task"]["lead"]["mode"] == "headless" and handoff is not None and handoff.status == "open" ) - role = "lead" if headless_lead else "reviewer" + delivery_implementer = ( + workflow == "issue-delivery" and "candidate" not in snapshot + ) + role = ( + "lead" if headless_lead + else "implementer" if delivery_implementer + else "reviewer" + ) workflow_role = ( - "internal_review_fixture" in snapshot - or "review_adapter" in snapshot - or "review_adapters" in snapshot + "internal_fake_delay" not in snapshot + and workflow in {"branch-review", "issue-delivery"} ) profile_index = None profile_id = None @@ -109,8 +116,16 @@ def main(argv=None): attempt_selection = candidates[profile_index] profile_id = attempt_selection["profile_id"] selected = attempt_selection["profile"] - adapter_key = "lead_adapter" if headless_lead else "review_adapter" - adapters_key = "lead_adapters" if headless_lead else "review_adapters" + adapter_key = { + "implementer": "implementation_adapter", + "reviewer": "review_adapter", + "lead": "lead_adapter", + }[role] + adapters_key = { + "implementer": "implementation_adapters", + "reviewer": "review_adapters", + "lead": "lead_adapters", + }[role] adapters = snapshot.get(adapters_key) adapter = ( adapters.get(profile_id) @@ -129,11 +144,21 @@ def main(argv=None): selected["account_pool_id"], "verified" if adapter else "unknown", ) - module = ( - ("devsquad.codex_lead_worker" if adapter else "devsquad.lead_worker") - if headless_lead - else ("devsquad.codex_review_worker" if adapter else "devsquad.review_worker") - ) + if role == "implementer": + module = ( + "devsquad.claude_delivery_worker" + if adapter else "devsquad.delivery_worker" + ) + elif role == "lead": + module = ( + "devsquad.codex_lead_worker" + if adapter else "devsquad.lead_worker" + ) + else: + module = ( + "devsquad.codex_review_worker" + if adapter else "devsquad.review_worker" + ) command = [sys.executable, "-P", "-m", module] worker_snapshot = json.loads(canonical_json(snapshot)) worker_snapshot["routing"]["roles"][role]["selected"] = attempt_selection @@ -151,8 +176,10 @@ def main(argv=None): args.run_id, ( f"lead-workflow-input-{handoff.sequence}-{profile_index}.json" - if headless_lead - else f"workflow-input-{profile_index}.json" + if headless_lead else + f"implementation-input-{profile_index}.json" + if role == "implementer" else + f"workflow-input-{profile_index}.json" ), canonical_json(worker_snapshot).encode(), ) @@ -207,7 +234,8 @@ def main(argv=None): Service(Path(args.database).parent).resume(args.run_id) except ConflictError: pass - elif current["state"] == "queued" and current["phase"] is None: + elif (current["state"] == "queued" and current["phase"] is None + and not (workflow == "issue-delivery" and role == "implementer")): try: Service(Path(args.database).parent).resume(args.run_id) except ConflictError: diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 54fea98..5b656bd 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -49,6 +49,7 @@ assert_clean_inputs, committed_regular_file, prepare_check_workspace, + prepare_delivery_workspace, prepare_review_workspace, repo_relative_config, resolve_commit, @@ -404,6 +405,7 @@ def _resolve_snapshot( run_id: str | None = None, internal_review_fixture: dict[str, Any] | None = None, internal_lead_fixture: dict[str, Any] | None = None, + internal_implementation_fixture: dict[str, Any] | None = None, capacity_in_flight: dict[str, int] | None = None, ) -> dict[str, Any]: repo = resolved_repo or Path(task["project"]["repo_path"]).resolve(strict=True) @@ -450,25 +452,48 @@ def _resolve_snapshot( ) if project_id is None or run_id is None: raise ContractError("public preflight requires run-owned workspace identity") - snapshot["workspace"] = prepare_review_workspace( - repo, - self.runtime, - project_id, - run_id, - base_oid, - target_oid, - scope_paths, - required_clean_paths=config_paths.values(), - ) - snapshot["check_workspace"] = prepare_check_workspace( - repo, - self.runtime, - project_id, - run_id, - target_oid, - scope_paths, - required_clean_paths=config_paths.values(), - ) + if task["workflow"] == "branch-review": + snapshot["workspace"] = prepare_review_workspace( + repo, + self.runtime, + project_id, + run_id, + base_oid, + target_oid, + scope_paths, + required_clean_paths=config_paths.values(), + ) + snapshot["check_workspace"] = prepare_check_workspace( + repo, + self.runtime, + project_id, + run_id, + target_oid, + scope_paths, + required_clean_paths=config_paths.values(), + ) + else: + snapshot["delivery_workspace"] = prepare_delivery_workspace( + repo, + self.runtime, + project_id, + run_id, + target_oid, + task["scope"]["read_paths"], + task["scope"]["write_paths"], + required_clean_paths=config_paths.values(), + ) + if internal_implementation_fixture is None: + raise CapabilityUnavailable( + "live issue-delivery implementer execution is not available yet" + ) + if (not isinstance(internal_implementation_fixture, dict) + or set(internal_implementation_fixture) + != {"writes", "delay_seconds"}): + raise ContractError("internal implementation fixture is invalid") + snapshot["internal_implementation_fixture"] = json.loads( + canonical_json(internal_implementation_fixture) + ) if internal_review_fixture is not None: if (not isinstance(internal_review_fixture, dict) or set(internal_review_fixture) @@ -513,6 +538,9 @@ def _continue_preparation( internal_delay = submitted.get("_internal_fake_delay") internal_review_fixture = submitted.get("_internal_review_fixture") internal_lead_fixture = submitted.get("_internal_lead_fixture") + internal_implementation_fixture = submitted.get( + "_internal_implementation_fixture" + ) store.validate_predecessor(run_id, fencing_token, supersedes_run_id) validated_supersedes_run_id = supersedes_run_id validate_task(task, require_existing_repo=True) @@ -533,9 +561,11 @@ def _continue_preparation( run_id=run_id, internal_review_fixture=internal_review_fixture, internal_lead_fixture=internal_lead_fixture, + internal_implementation_fixture=internal_implementation_fixture, capacity_in_flight=store.active_pool_counts(), ) - if internal_delay is None and internal_review_fixture is None: + if (task["workflow"] == "branch-review" and internal_delay is None + and internal_review_fixture is None): reviewer_route = snapshot["routing"]["roles"]["reviewer"] reviewer_candidates = [ reviewer_route["selected"], *reviewer_route["fallbacks"], @@ -568,7 +598,9 @@ def _continue_preparation( package_path=str(package), package_digest=digest, supersedes_run_id=supersedes_run_id, - worktree_path=(snapshot.get("workspace") or {}).get("path"), + worktree_path=( + snapshot.get("workspace") or snapshot.get("delivery_workspace") or {} + ).get("path"), ) return (version, package, digest), None except (BudgetExhausted, CapabilityUnavailable, ProfileUnsupported) as exc: @@ -615,12 +647,17 @@ def start( _internal_fake_delay: float | None = None, _internal_review_fixture: dict[str, Any] | None = None, _internal_lead_fixture: dict[str, Any] | None = None, + _internal_implementation_fixture: dict[str, Any] | None = None, ) -> dict[str, Any]: validate_task(task, require_existing_repo=True) if _internal_fake_delay is not None and _internal_review_fixture is not None: raise ContractError("internal lifecycle fixtures are mutually exclusive") if _internal_lead_fixture is not None and _internal_fake_delay is not None: raise ContractError("internal lifecycle fixtures are mutually exclusive") + if (_internal_implementation_fixture is not None + and (_internal_fake_delay is not None + or task["workflow"] != "issue-delivery")): + raise ContractError("internal implementation fixture requires issue-delivery") submitted = {"task": task, "supersedes_run_id": supersedes_run_id} if _internal_fake_delay is not None: submitted["_internal_fake_delay"] = _internal_fake_delay @@ -628,6 +665,10 @@ def start( submitted["_internal_review_fixture"] = _internal_review_fixture if _internal_lead_fixture is not None: submitted["_internal_lead_fixture"] = _internal_lead_fixture + if _internal_implementation_fixture is not None: + submitted["_internal_implementation_fixture"] = ( + _internal_implementation_fixture + ) store = self._store() try: claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}") @@ -673,6 +714,17 @@ def status(self, run_id: str) -> dict[str, Any]: next_action = "continue_headless_lead" if headless else "claim_handoff" elif run["state"] == "awaiting_host": next_action = "handoff_submission_saved" + elif run["state"] == "queued" and run["phase"] is None: + try: + snapshot = self._review_snapshot(run) + next_action = ( + "resume_candidate_review" + if snapshot.get("task", {}).get("workflow") == "issue-delivery" + and isinstance(snapshot.get("candidate"), dict) + else None + ) + except ConflictError: + next_action = None else: next_action = None return { diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 9f0721d..01ee311 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1208,6 +1208,107 @@ def commit_durable_fallback( self.connection.execute("ROLLBACK") raise + def commit_delivery_candidate( + self, + run_id: str, + attempt_token: str, + artifacts: list[dict[str, Any]], + metadata: Any, + mutable_snapshot: dict[str, Any], + review_worktree_path: str, + candidate: dict[str, Any], + ) -> str: + """Atomically import one implementation and queue its frozen candidate.""" + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( + run_id, artifacts, require_result_receipt=False, + ) + encoded_metadata = canonical_json(metadata) + encoded_snapshot = canonical_json(mutable_snapshot) + if (not isinstance(candidate, dict) + or mutable_snapshot.get("candidate") != candidate): + raise ContractError("delivery candidate differs from its saved snapshot") + expected_lengths = { + "candidate_sha256": 64, + "commit_oid": 40, + "patch_sha256": 64, + } + for field, expected_length in expected_lengths.items(): + value = candidate.get(field) + if (not isinstance(value, str) or len(value) != expected_length + or any(character not in "0123456789abcdef" for character in value)): + raise ContractError(f"delivery candidate {field} is invalid") + try: + resolved_worktree = str(Path(review_worktree_path).resolve(strict=True)) + worktree_common = str(git_common_dir(Path(resolved_worktree))) + except OSError as exc: + raise ContractError("delivery review worktree is unavailable") from exc + + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT r.state,r.phase,r.version,r.mutable_snapshot,p.git_common_dir " + "FROM runs r JOIN projects p ON p.id=r.project_id WHERE r.id=?", + (run_id,), + ).fetchone() + attempt = self.connection.execute( + "SELECT id,status,role,stdout_artifact_id,stderr_artifact_id," + "output_metadata FROM attempts WHERE run_id=? AND attempt_token=?", + (run_id, attempt_token), + ).fetchone() + if not run or not attempt: + raise ConflictError("delivery candidate import is fenced") + if (attempt["status"] == "finished" and run["state"] == "queued" + and run["phase"] is None + and run["mutable_snapshot"] == encoded_snapshot): + self.connection.execute("COMMIT") + return "candidate_ready" + if (attempt["status"] != "running" or attempt["role"] != "implementer" + or run["state"] != "running" or run["phase"] is not None): + raise ConflictError("delivery candidate import is fenced") + if worktree_common != run["git_common_dir"]: + raise ContractError("delivery review worktree belongs to another project") + version, artifact_ids = self._reference_prepared_artifacts( + run_id, run["version"], prepared, + ) + version = self._record_prepared_output( + run_id, + version, + attempt, + artifact_ids, + stdout_name, + stderr_name, + encoded_metadata, + ) + now, version = _utc_now(), version + 1 + self.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", + (now, attempt["id"]), + ) + self.connection.execute( + "UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,), + ) + self.connection.execute( + "UPDATE runs SET mutable_snapshot=?,worktree_path=?,state='queued'," + "phase=NULL,version=?,updated_at=? WHERE id=?", + (encoded_snapshot, resolved_worktree, version, now, run_id), + ) + payload = canonical_json({ + "attempt_id": attempt["id"], + "candidate_sha256": candidate["candidate_sha256"], + "commit_oid": candidate["commit_oid"], + "patch_sha256": candidate["patch_sha256"], + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'delivery.candidate_ready',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return "candidate_ready" + except Exception: + self.connection.execute("ROLLBACK") + raise + def commit_durable_handoff( self, run_id: str, diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 158be7b..0665ac5 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -20,7 +20,17 @@ from .contracts import ContractError, LaunchSpec from .reports import build_early_terminal_reports from .store import AttemptReservation, ConflictError, Store, canonical_json -from .workflows import decode_branch_review_evidence, decode_headless_lead_evidence +from .workflows import ( + decode_branch_review_evidence, + decode_headless_lead_evidence, + validate_implementation_evidence, +) +from .workspaces import ( + freeze_delivery_candidate, + prepare_check_workspace, + prepare_review_workspace, + repo_relative_config, +) def _open_stdin_artifact(path: str) -> BinaryIO: @@ -74,6 +84,11 @@ def inspect_process(pid: int, pgid: int, expected_start: str) -> str: return "dead" except PermissionError: return "ambiguous" + try: + if not _live_group_exists(pgid): + return "dead" + except RuntimeError: + pass return "ambiguous" try: observed_pgid = os.getpgid(pid) @@ -372,6 +387,112 @@ def _commit_review_handoff( packet, ) + def _commit_delivery_candidate( + self, + run_id: str, + attempt: dict[str, Any], + stream_artifacts: list[dict[str, Any]], + metadata: dict[str, Any], + snapshot: dict[str, Any], + stdout: bytes, + ) -> str: + evidence = validate_implementation_evidence( + json.loads(stdout.decode("utf-8")), snapshot, + ) + task = snapshot["task"] + delivery = snapshot["delivery_workspace"] + source_repo = Path(task["project"]["repo_path"]).resolve(strict=True) + workspace = Path(delivery["path"]).resolve(strict=True) + candidate, patch = freeze_delivery_candidate( + source_repo, + workspace, + delivery["baseline_oid"], + task["scope"]["write_paths"], + run_id, + ) + project_id = self.store.run(run_id)["project_id"] + scope_paths = tuple(dict.fromkeys( + task["scope"]["read_paths"] + task["scope"]["write_paths"] + )) + config_paths = tuple( + repo_relative_config(source_repo, task["routing"][label], label) + for label in ("profiles_file", "policy_file") + ) + review_workspace = prepare_review_workspace( + source_repo, + self.store.database.parent, + project_id, + run_id, + delivery["baseline_oid"], + candidate["commit_oid"], + scope_paths, + required_clean_paths=config_paths, + candidate_sha256=candidate["candidate_sha256"], + ) + check_workspace = prepare_check_workspace( + source_repo, + self.store.database.parent, + project_id, + run_id, + candidate["commit_oid"], + scope_paths, + required_clean_paths=config_paths, + ) + iteration = len(snapshot.get("delivery_iterations", [])) + 1 + candidate_record = { + **candidate, + "iteration": iteration, + "patch_artifact": f"candidate-{iteration}.patch", + "implementation_artifact": ( + f"implementation-attempt-{attempt['id']}.json" + ), + } + new_snapshot = json.loads(canonical_json(snapshot)) + new_snapshot["candidate"] = candidate_record + new_snapshot["workspace"] = review_workspace + new_snapshot["check_workspace"] = check_workspace + iterations = list(new_snapshot.get("delivery_iterations", [])) + iterations.append({ + "iteration": iteration, + "candidate": candidate_record, + "implementation": evidence, + }) + new_snapshot["delivery_iterations"] = iterations + + artifacts = list(stream_artifacts) + documents = { + f"candidate-{iteration}.json": candidate_record, + f"implementation-attempt-{attempt['id']}.json": evidence, + } + for name, document in documents.items(): + content = (canonical_json(document) + "\n").encode() + path, digest, size = self.store.finalize_artifact( + run_id, name, content, + ) + artifacts.append({ + "name": name, "path": path, "sha256": digest, + "byte_size": size, + }) + patch_name = f"candidate-{iteration}.patch" + path, digest, size = self.store.finalize_artifact( + run_id, patch_name, patch, + ) + if digest != candidate["patch_sha256"] or size != candidate["patch_bytes"]: + raise ContractError("saved candidate patch differs from its identity") + artifacts.append({ + "name": patch_name, "path": path, "sha256": digest, + "byte_size": size, + }) + return self.store.commit_delivery_candidate( + run_id, + attempt["attempt_token"], + artifacts, + metadata, + new_snapshot, + review_workspace["path"], + candidate_record, + ) + def import_durable(self, run_id: str) -> str: attempt=self.store.attempt(run_id) if not attempt or attempt["status"] not in {"running","cancelling"}: @@ -431,14 +552,29 @@ def import_durable(self, run_id: str) -> str: path,digest,size=self.store.finalize_artifact(run_id,logical,data) artifacts.append({"name":logical,"path":path,"sha256":digest,"byte_size":size}) snapshot=json.loads(self.store.run(run_id)["mutable_snapshot"]) - workflow_review = ( - "internal_review_fixture" in snapshot - or "review_adapter" in snapshot - or "review_adapters" in snapshot - ) + workflow = snapshot.get("task", {}).get("workflow") + managed_workflow = "internal_fake_delay" not in snapshot + workflow_review = managed_workflow and workflow == "branch-review" + workflow_delivery = managed_workflow and workflow == "issue-delivery" role = attempt.get("role", "worker") semantic_error=None - if (role == "lead" and workflow_review and not receipt["cancelled"] + if (role == "implementer" and workflow_delivery + and not receipt["cancelled"] + and not receipt["timed_out"] and receipt["returncode"] == 0): + try: + return self._commit_delivery_candidate( + run_id, + attempt, + artifacts, + metadata, + snapshot, + captures["stdout"], + ) + except (ContractError, UnicodeDecodeError, json.JSONDecodeError) as exc: + semantic_error = str(exc) + receipt["error"] = "IMPLEMENTATION_OUTPUT_INVALID" + receipt["message"] = semantic_error + elif (role == "lead" and workflow_review and not receipt["cancelled"] and not receipt["timed_out"] and receipt["returncode"]==0): try: handoff = self.store.handoff_snapshot(run_id) diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index d03f084..c2fd3be 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -484,6 +484,144 @@ def _frozen_attempt_selection( raise ContractError(f"{role} attempt is outside the frozen fallback set") +def build_implementation_prompt( + task: dict[str, Any], + delivery_workspace: dict[str, Any], +) -> str: + """Build one bounded implementation assignment from frozen host input.""" + validate_task(task) + if task["workflow"] != "issue-delivery": + raise ContractError("implementation prompt requires an issue-delivery task") + if not isinstance(delivery_workspace, dict): + raise ContractError("delivery workspace snapshot must be an object") + baseline_oid = _commit_oid( + delivery_workspace.get("baseline_oid"), "delivery baseline_oid", + ) + if delivery_workspace.get("write_scope") != task["scope"]["write_paths"]: + raise ContractError("delivery write scope differs from the frozen task") + assignment = { + "goal": task["goal"], + "acceptance": task["acceptance"], + "read_scope": task["scope"]["read_paths"], + "write_scope": task["scope"]["write_paths"], + "baseline_oid": baseline_oid, + } + return "\n".join([ + "You are the sole implementation writer for one bounded Git task.", + "Edit only the declared write scope in the supplied isolated worktree.", + "Do not commit, merge, push, publish, delegate, or change remotes.", + "Do not run checks; the coordinator runs declared checks separately.", + "When the edits are complete, return a concise implementation summary.", + "Frozen assignment:", + canonical_json(assignment), + ]) + + +def validate_implementation_evidence( + value: dict[str, Any], + snapshot: dict[str, Any], +) -> dict[str, Any]: + """Validate a completed writer attempt before candidate freezing.""" + document = _exact(value, { + "schema_version", "workflow", "baseline_oid", "summary", "attempt", + }, "implementation evidence") + if document["schema_version"] != 1 or type(document["schema_version"]) is not int: + raise ContractError("implementation evidence schema_version is invalid") + if document["workflow"] != "issue-delivery": + raise ContractError("implementation evidence workflow is invalid") + if not isinstance(snapshot, dict): + raise ContractError("frozen delivery snapshot is invalid") + task = snapshot.get("task") + workspace = snapshot.get("delivery_workspace") + if not isinstance(task, dict) or not isinstance(workspace, dict): + raise ContractError("frozen delivery snapshot is incomplete") + if task.get("workflow") != "issue-delivery": + raise ContractError("implementation evidence requires issue-delivery") + if _commit_oid( + document["baseline_oid"], "implementation baseline_oid", + ) != workspace.get("baseline_oid"): + raise ContractError("implementation evidence targets a different baseline") + _text(document["summary"], "implementation summary") + attempt = _exact(document["attempt"], { + "role", "selected_profile", "prompt_sha256", "observed_identity", + "native_ids", "worker_invocations", "native_model_requests", "usage", + }, "implementation attempt evidence") + if attempt["role"] != "implementer": + raise ContractError("implementation attempt role is invalid") + selected = _frozen_attempt_selection( + snapshot, "implementer", attempt["selected_profile"], + ) + prompt_sha256 = hashlib.sha256( + build_implementation_prompt(task, workspace).encode() + ).hexdigest() + if _sha256( + attempt["prompt_sha256"], "implementation prompt sha256", + ) != prompt_sha256: + raise ContractError("implementation prompt hash does not match the task") + adapter = snapshot.get("implementation_adapters", {}).get( + selected["profile_id"] + ) + if adapter is None: + if attempt["observed_identity"] is not None or attempt["native_ids"] != {}: + raise ContractError("fixture implementation cannot claim native identity") + elif not isinstance(attempt["observed_identity"], dict): + raise ContractError("native implementation identity is missing") + if attempt["worker_invocations"] != 1 or type(attempt["worker_invocations"]) is not int: + raise ContractError("implementation worker invocation accounting is invalid") + native_requests = attempt["native_model_requests"] + if native_requests is not None and ( + type(native_requests) is not int or native_requests < 0): + raise ContractError("implementation native request count is invalid") + usage = _exact(attempt["usage"], { + "input_tokens", "output_tokens", "total_tokens", "source", + }, "implementation usage") + for field in ("input_tokens", "output_tokens", "total_tokens"): + if usage[field] is not None and ( + type(usage[field]) is not int or usage[field] < 0): + raise ContractError("implementation token usage is invalid") + if usage["source"] not in {"native_reported", "unavailable"}: + raise ContractError("implementation usage source is invalid") + if usage["source"] == "unavailable" and any( + usage[field] is not None + for field in ("input_tokens", "output_tokens", "total_tokens") + ): + raise ContractError("unavailable implementation usage cannot invent tokens") + return json.loads(canonical_json(document)) + + +def make_implementation_evidence( + snapshot: dict[str, Any], + summary: str, +) -> dict[str, Any]: + selected = snapshot["routing"]["roles"]["implementer"]["selected"] + document = { + "schema_version": 1, + "workflow": "issue-delivery", + "baseline_oid": snapshot["delivery_workspace"]["baseline_oid"], + "summary": summary, + "attempt": { + "role": "implementer", + "selected_profile": selected, + "prompt_sha256": hashlib.sha256( + build_implementation_prompt( + snapshot["task"], snapshot["delivery_workspace"], + ).encode() + ).hexdigest(), + "observed_identity": None, + "native_ids": {}, + "worker_invocations": 1, + "native_model_requests": None, + "usage": { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + }, + }, + } + return validate_implementation_evidence(document, snapshot) + + def _frozen_role_adapter( snapshot: dict[str, Any], role: str, diff --git a/plugin/core/src/devsquad/workspaces.py b/plugin/core/src/devsquad/workspaces.py index 0caac8c..37e14c8 100644 --- a/plugin/core/src/devsquad/workspaces.py +++ b/plugin/core/src/devsquad/workspaces.py @@ -3,6 +3,7 @@ from __future__ import annotations import hashlib +import fcntl import os from pathlib import Path, PurePosixPath import subprocess @@ -271,6 +272,7 @@ def prepare_review_workspace( scope_paths: Iterable[str], *, required_clean_paths: Iterable[str] = (), + candidate_sha256: str | None = None, ) -> dict[str, object]: """Create or validate one detached, run-owned worktree at the target commit.""" repo = source_repo.resolve(strict=True) @@ -292,10 +294,17 @@ def prepare_review_workspace( "target_oid": target_oid, "changed_paths": sorted(changed), } + computed_candidate = hashlib.sha256(canonical_json(identity).encode()).hexdigest() + if candidate_sha256 is not None: + if (not isinstance(candidate_sha256, str) + or len(candidate_sha256) != 64 + or any(character not in "0123456789abcdef" for character in candidate_sha256)): + raise ContractError("candidate SHA-256 override is invalid") + computed_candidate = candidate_sha256 return { **identity, "path": str(workspace.resolve()), - "candidate_sha256": hashlib.sha256(canonical_json(identity).encode()).hexdigest(), + "candidate_sha256": computed_candidate, "scope": list(scopes), } @@ -450,7 +459,7 @@ def _candidate_snapshot( }, patch -def freeze_delivery_candidate( +def _freeze_delivery_candidate_unlocked( source_repo: Path, workspace: Path, baseline_oid: str, @@ -529,3 +538,24 @@ def freeze_delivery_candidate( if dirty_paths(delivery): raise ContractError("delivery workspace remained dirty after candidate commit") return snapshot, patch + + +def freeze_delivery_candidate( + source_repo: Path, + workspace: Path, + baseline_oid: str, + write_paths: Iterable[str], + run_id: str, +) -> tuple[dict[str, object], bytes]: + """Serialize candidate freezing so competing recovery importers replay it.""" + delivery = workspace.resolve(strict=True) + lock_path = delivery.parent / ".candidate-finalize.lock" + descriptor = os.open(lock_path, os.O_RDWR | os.O_CREAT, 0o600) + try: + fcntl.flock(descriptor, fcntl.LOCK_EX) + return _freeze_delivery_candidate_unlocked( + source_repo, delivery, baseline_oid, write_paths, run_id, + ) + finally: + fcntl.flock(descriptor, fcntl.LOCK_UN) + os.close(descriptor) diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py index 4b48bbd..e490ffb 100644 --- a/test/core/test_delivery_workflow.py +++ b/test/core/test_delivery_workflow.py @@ -1,9 +1,11 @@ from __future__ import annotations import hashlib +import json import subprocess import sys import tempfile +import time import unittest from pathlib import Path @@ -11,6 +13,8 @@ sys.path.insert(0, str(CORE / "src")) from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.store import Store from devsquad.workspaces import ( freeze_delivery_candidate, prepare_delivery_workspace, @@ -35,6 +39,14 @@ def setUp(self): (self.repo / "src/app.py").write_text("VALUE = 'base'\n") (self.repo / "tests/test_app.py").write_text("# base test\n") (self.repo / "README.md").write_text("fixture\n") + (self.repo / "devsquad").mkdir() + profiles, policy = self.delivery_routing_documents() + (self.repo / "devsquad/profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n" + ) + (self.repo / "devsquad/policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n" + ) self.git(self.repo, "add", ".") self.git(self.repo, "commit", "-qm", "base") self.baseline = resolve_commit(self.repo, "HEAD") @@ -47,6 +59,78 @@ def setUp(self): self.source_refs = self.git(self.repo, "show-ref") self.remote_refs = self.git(self.remote, "show-ref") + @staticmethod + def delivery_routing_documents(): + def profile( + profile_id: str, + *, + family: str, + model: str, + permission: str, + ) -> dict[str, object]: + return { + "id": profile_id, + "harness": "fixture", + "model_family": family, + "model_id": model, + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read", "write"] if permission == "workspace_write" else ["read"], + "permission_policy": permission, + "account_pool_id": f"{profile_id}-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["tracked-fixture"], + } + + profiles = { + "schema_version": 1, + "profiles": [ + profile( + "fixture-implementer", + family="fixture-family-a", + model="fixture-write-model", + permission="workspace_write", + ), + profile( + "fixture-reviewer", + family="fixture-family-b", + model="fixture-review-model", + permission="read_only", + ), + ], + "bindings": {}, + } + policy = { + "schema_version": 1, + "id": "delivery-fixture-policy", + "version": 1, + "roles": { + "implementer": [ + {"kind": "profile", "id": "fixture-implementer"} + ], + "reviewer": [ + {"kind": "profile", "id": "fixture-reviewer"} + ], + }, + "task_classes": {"fixture-delivery-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": { + "fixture-implementer-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + "fixture-reviewer-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + return profiles, policy + @staticmethod def git(repo: Path, *args: str) -> str: return subprocess.run( @@ -68,6 +152,80 @@ def prepare(self) -> dict[str, object]: ("src/app.py", "tests"), ) + def delivery_task(self) -> dict[str, object]: + return { + "schema_version": 1, + "project": { + "repo_path": str(self.repo), + "base_ref": self.baseline, + "target_ref": self.baseline, + }, + "workflow": "issue-delivery", + "goal": "Fix the fixture value and add one regression test.", + "task_class": "fixture-delivery-small", + "acceptance": [ + { + "id": "value-fixed", + "description": "The fixture value is fixed.", + "evidence_kind": "check", + }, + { + "id": "independent-review", + "description": "A different model reviews the candidate.", + "evidence_kind": "review", + }, + ], + "checks": [ + { + "id": "fixture-check", + "argv": ["python3", "-c", "print('ok')"], + "cwd": ".", + "timeout_seconds": 10, + "required_to_pass": True, + } + ], + "scope": { + "read_paths": ["src", "tests"], + "write_paths": ["src/app.py", "tests"], + }, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": 30, + "max_worker_invocations": 5, + "max_revisions": 1, + "max_fallbacks_per_step": 0, + }, + "origin": {"surface": "test"}, + } + + @staticmethod + def implementation_fixture(delay: float = 0) -> dict[str, object]: + return { + "writes": [ + {"path": "src/app.py", "content": "VALUE = 'fixed'\n"}, + { + "path": "tests/test regression.py", + "content": "def test_regression():\n assert True\n", + }, + ], + "delay_seconds": delay, + } + + def wait_for_candidate(self, service: Service, run_id: str) -> dict[str, object]: + deadline = time.monotonic() + 15 + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] == "queued" and status["next_action"] == "resume_candidate_review": + return status + if status["state"] in {"failed", "cancelled"}: + self.fail(f"delivery run terminalized early: {status}") + time.sleep(0.05) + self.fail(f"delivery candidate did not become ready: {service.status(run_id)}") + def assert_source_unchanged(self) -> None: self.assertEqual(resolve_commit(self.repo, "HEAD"), self.baseline) self.assertEqual(self.git(self.repo, "status", "--porcelain"), self.source_status) @@ -178,6 +336,104 @@ def test_candidate_requires_a_change(self): ) self.assert_source_unchanged() + def test_durable_implementer_publishes_candidate_artifacts_once(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "durable-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + ) + status = self.wait_for_candidate(service, started["run_id"]) + self.assertEqual(status["phase"], None) + + store = Store(service.database, service.artifacts) + self.addCleanup(store.close) + run = store.run(started["run_id"]) + snapshot = json.loads(run["mutable_snapshot"]) + candidate = snapshot["candidate"] + attempts = store.attempts_for_run(started["run_id"]) + artifacts = {item["name"]: item for item in store.artifacts_for_run(started["run_id"])} + self.assertEqual(len(attempts), 1) + self.assertEqual(attempts[0]["role"], "implementer") + self.assertEqual(attempts[0]["status"], "finished") + self.assertEqual(store.worker_invocations(started["run_id"]), 1) + self.assertIn("candidate-1.json", artifacts) + self.assertIn("candidate-1.patch", artifacts) + self.assertIn( + f"implementation-attempt-{attempts[0]['id']}.json", artifacts, + ) + self.assertEqual( + artifacts["candidate-1.patch"]["sha256"], candidate["patch_sha256"], + ) + self.assertEqual( + resolve_commit(Path(snapshot["workspace"]["path"]), "HEAD"), + candidate["commit_oid"], + ) + self.assertEqual( + resolve_commit(Path(snapshot["check_workspace"]["path"]), "HEAD"), + candidate["commit_oid"], + ) + event_types = [ + event["type"] for event in store.events_for_run(started["run_id"]) + ] + self.assertEqual(event_types.count("delivery.candidate_ready"), 1) + self.assert_source_unchanged() + + def test_live_implementer_cannot_be_resumed_into_a_second_writer(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "one-writer-delivery", + _internal_implementation_fixture=self.implementation_fixture(0.5), + ) + deadline = time.monotonic() + 10 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "running": + break + time.sleep(0.02) + else: + self.fail("implementer did not enter running state") + resumed = service.resume(started["run_id"]) + self.assertEqual(resumed["disposition"], "live") + self.assertFalse(resumed["launched"]) + self.wait_for_candidate(service, started["run_id"]) + store = Store(service.database, service.artifacts) + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual(len(attempts), 1) + self.assertEqual(store.worker_invocations(started["run_id"]), 1) + finally: + store.close() + self.assert_source_unchanged() + + def test_durable_out_of_scope_implementation_cannot_publish_a_candidate(self): + service = Service(self.runtime) + fixture = { + "writes": [{"path": "README.md", "content": "unauthorized\n"}], + "delay_seconds": 0, + } + started = service.start( + self.delivery_task(), + "out-of-scope-delivery", + _internal_implementation_fixture=fixture, + ) + deadline = time.monotonic() + 15 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "failed": + break + time.sleep(0.05) + else: + self.fail("out-of-scope delivery did not fail") + result = service.result(started["run_id"]) + self.assertTrue(result["ready"]) + self.assertEqual(result["state"], "failed") + self.assertNotIn( + "candidate-1.json", {item["name"] for item in result["artifacts"]}, + ) + self.assert_source_unchanged() + if __name__ == "__main__": unittest.main() diff --git a/test/core/test_supervisor.py b/test/core/test_supervisor.py index ee3f0a3..ae827b3 100644 --- a/test/core/test_supervisor.py +++ b/test/core/test_supervisor.py @@ -62,6 +62,20 @@ def test_process_inventory_failure_never_means_absence(self): with mock.patch("devsquad.supervisor.subprocess.run", return_value=failed), self.assertRaises(RuntimeError): _live_group_exists(123) + def test_zombie_only_group_is_dead_but_live_unknown_group_is_ambiguous(self): + with ( + mock.patch("devsquad.supervisor.process_start_identity", return_value=None), + mock.patch("devsquad.supervisor.os.killpg"), + mock.patch("devsquad.supervisor._live_group_exists", return_value=False), + ): + self.assertEqual(inspect_process(123, 123, "expected"), "dead") + with ( + mock.patch("devsquad.supervisor.process_start_identity", return_value=None), + mock.patch("devsquad.supervisor.os.killpg"), + mock.patch("devsquad.supervisor._live_group_exists", return_value=True), + ): + self.assertEqual(inspect_process(123, 123, "expected"), "ambiguous") + def test_normal_and_fast_completion_persist_bounded_output(self): run_id, handle = self.launch("output", "import sys; print('o'*100); print('e'*100,file=sys.stderr)") self.assertEqual(self.supervisor.wait(handle, 3), 0) From 64bcb2c9db165a62cfe993280aaef1f9a0bfd06f Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 14:57:02 +0200 Subject: [PATCH 084/197] docs: complete M5 delivery foundation plan --- docs/plans/engineering-team/M5-STATUS.md | 23 +++++++++++++++++++++-- docs/plans/engineering-team/RESUME.md | 16 ++++++++++++---- docs/plans/engineering-team/backlog.json | 9 +++++++++ 3 files changed, 42 insertions(+), 6 deletions(-) diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index 84fdc09..ebfb544 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -8,8 +8,8 @@ two-harness gate passes. | Requirement | Planned evidence | Status | |---|---|---| | Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | verified offline at `d96e9e4` | -| Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | workspace isolation verified offline at `6a7e849`; workflow writer pending | -| Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `6a7e849` | +| Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | verified offline at `0e88d73` | +| Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `0e88d73` | | Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | router verified; workflow pending | | Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | pending | | Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | pending | @@ -52,3 +52,22 @@ assertions. The first full discovery observed the pre-existing coordinator- crash race test return `live` once; that isolated test and the full discovery rerun passed unchanged. Connecting this workspace to the durable implementer attempt and its existing store-level writer fence remains the next slice. + +## Plan 07-01 checkpoint 3 + +The offline implementation worker now runs inside the durable M2 process and +writer fence. Its frozen prompt/profile evidence must validate before the +coordinator serializes candidate finalization, rechecks scope, commits locally, +creates independent review/check worktrees and atomically imports the +implementation, candidate and patch artifacts into the same saved ledger. +A concurrent resume observes the live owner and creates no second attempt. +An out-of-scope implementation terminalizes failed without publishing a +candidate. The original checkout and remote refs remain unchanged. + +The complete gate is 224 core tests (2 optional-SDK skips) and 220 Bash +assertions. During this slice a full run reproduced an existing macOS race in +which a zombie-only process group was classified `ambiguous`; `0e88d73` now +uses the non-zombie process-group inventory before declaring ambiguity and has +a direct regression. The isolated service/supervisor/delivery suites and the +complete discovery pass after that fix. Plan 07-02 now owns independent review, +checks, disposition and revision behavior. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 83d3f42..af3a7c0 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -94,6 +94,14 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne tests and 220 Bash assertions. One first core run observed the pre-existing coordinator-crash test return `live`; its isolated run and the complete rerun passed unchanged. +- M5 Plan 07-01 is complete at `0e88d73`. The offline implementer now executes + inside the durable process/writer fence, validates frozen attempt evidence, + serializes candidate finalization across recovery importers and atomically + saves implementation, local commit and patch evidence before creating + separate review/check worktrees. Concurrent resume creates no second writer; + out-of-scope edits fail without a candidate. The complete gate is 224 core + tests and 220 Bash assertions. A reproduced macOS zombie-only process-group + ambiguity was fixed with a non-zombie inventory check and direct regression. ## Completed and preserved @@ -105,7 +113,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 220 tests passed through M5 Plan 07-01 delivery candidate slice with warnings promoted to errors | +| Python core discovery | 224 tests passed through completed M5 Plan 07-01 with warnings promoted to errors | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -142,9 +150,9 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M5 Plan 07-01 by connecting the isolated delivery worktree to a - durable fenced implementer attempt, then publish its local candidate/patch - artifacts to the saved run without merge, push or publication. +2. Execute M5 Plan 07-02: generalize candidate-bound review/check evidence for + `issue-delivery`, publish the lead handoff, and make a bounded `revise` + disposition return to the fenced implementer with all iterations retained. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 0e9fe93..4a95c0d 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -230,6 +230,15 @@ "artifact": "M5-STATUS.md", "recorded_at": "2026-09-23T18:00:27+05:30", "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "0e88d73", + "command_or_action": "224 core tests with ResourceWarning promoted to error, 220 Bash assertions and focused delivery/service/supervisor suites", + "outcome": "The durable implementer is fenced to one writer, validates frozen attempt evidence, serializes recovery-time candidate freezing and atomically publishes local candidate/patch artifacts plus separate review/check worktrees; invalid scope cannot publish a candidate and the source checkout/remote remain unchanged", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-23T18:37:00+05:30", + "availability": "tracked_tests" } ], "blocker": null From 30b98df2f62358f9ce373d4a3b1445e9fea2d59e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Wed, 23 Sep 2026 23:19:41 -0700 Subject: [PATCH 085/197] feat: review frozen delivery candidates --- plugin/core/src/devsquad/service.py | 32 ++++++++------ plugin/core/src/devsquad/supervisor.py | 20 ++++++++- plugin/core/src/devsquad/workflows.py | 15 +++---- test/core/test_delivery_workflow.py | 58 ++++++++++++++++++++++++++ test/core/test_review_workflow.py | 22 ++++++++++ 5 files changed, 126 insertions(+), 21 deletions(-) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 5b656bd..1b88202 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -499,17 +499,22 @@ def _resolve_snapshot( or set(internal_review_fixture) != {"verdict", "summary", "findings"}): raise ContractError("internal review fixture fields are invalid") - fixture_document = { - "schema_version": 1, - "candidate_sha256": snapshot["workspace"]["candidate_sha256"], - "base_oid": base_oid, - "target_oid": target_oid, - "review_mode": review_mode(task), - **internal_review_fixture, - } - snapshot["internal_review_fixture"] = validate_review_document( - fixture_document, task, snapshot["workspace"], - ) + if task["workflow"] == "branch-review": + fixture_document = { + "schema_version": 1, + "candidate_sha256": snapshot["workspace"]["candidate_sha256"], + "base_oid": base_oid, + "target_oid": target_oid, + "review_mode": review_mode(task), + **internal_review_fixture, + } + snapshot["internal_review_fixture"] = validate_review_document( + fixture_document, task, snapshot["workspace"], + ) + else: + snapshot["pending_review_fixture"] = json.loads( + canonical_json(internal_review_fixture) + ) if internal_lead_fixture is not None: if (task["lead"]["mode"] != "headless" or not isinstance(internal_lead_fixture, dict) @@ -564,7 +569,10 @@ def _continue_preparation( internal_implementation_fixture=internal_implementation_fixture, capacity_in_flight=store.active_pool_counts(), ) - if (task["workflow"] == "branch-review" and internal_delay is None + if ((task["workflow"] == "branch-review" + or (task["workflow"] == "issue-delivery" + and internal_implementation_fixture is None)) + and internal_delay is None and internal_review_fixture is None): reviewer_route = snapshot["routing"]["roles"]["reviewer"] reviewer_candidates = [ diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 0665ac5..5ab120d 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -23,7 +23,9 @@ from .workflows import ( decode_branch_review_evidence, decode_headless_lead_evidence, + review_mode, validate_implementation_evidence, + validate_review_document, ) from .workspaces import ( freeze_delivery_candidate, @@ -364,7 +366,7 @@ def _commit_review_handoff( evidence_references.append({"name": name, "sha256": digest}) packet = { "schema_version": 1, - "workflow": "branch-review", + "workflow": evidence["workflow"], "candidate_sha256": evidence["candidate_sha256"], "base_oid": evidence["base_oid"], "target_oid": evidence["target_oid"], @@ -451,6 +453,19 @@ def _commit_delivery_candidate( new_snapshot["candidate"] = candidate_record new_snapshot["workspace"] = review_workspace new_snapshot["check_workspace"] = check_workspace + pending_review = new_snapshot.pop("pending_review_fixture", None) + if pending_review is not None: + fixture_document = { + "schema_version": 1, + "candidate_sha256": review_workspace["candidate_sha256"], + "base_oid": review_workspace["base_oid"], + "target_oid": review_workspace["target_oid"], + "review_mode": review_mode(task), + **pending_review, + } + new_snapshot["internal_review_fixture"] = validate_review_document( + fixture_document, task, review_workspace, + ) iterations = list(new_snapshot.get("delivery_iterations", [])) iterations.append({ "iteration": iteration, @@ -601,7 +616,8 @@ def import_durable(self, run_id: str) -> str: semantic_error=str(exc) receipt["error"]="HEADLESS_LEAD_OUTPUT_INVALID" receipt["message"]=semantic_error - elif (workflow_review and not receipt["cancelled"] + elif (role == "reviewer" and (workflow_review or workflow_delivery) + and not receipt["cancelled"] and not receipt["timed_out"] and receipt["returncode"]==0): try: return self._commit_review_handoff( diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index c2fd3be..b997de6 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -211,8 +211,8 @@ def validate_review_document( ) -> dict[str, Any]: """Validate model output and bind it to the exact frozen candidate.""" validate_task(task) - if task["workflow"] != "branch-review": - raise ContractError("review document requires a branch-review task") + if task["workflow"] not in {"branch-review", "issue-delivery"}: + raise ContractError("review document requires a reviewable workflow") document = _exact(value, { "schema_version", "candidate_sha256", "base_oid", "target_oid", "review_mode", "verdict", "summary", "findings", @@ -427,8 +427,8 @@ def apply_lead_disposition( def build_review_prompt(task: dict[str, Any], workspace: dict[str, Any]) -> str: """Build a deterministic, read-only prompt from host-authorized fields only.""" validate_task(task) - if task["workflow"] != "branch-review": - raise ContractError("review prompt requires a branch-review task") + if task["workflow"] not in {"branch-review", "issue-delivery"}: + raise ContractError("review prompt requires a reviewable workflow") candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) mode = review_mode(task) focus = task.get("review", {}).get("focus") @@ -868,13 +868,14 @@ def validate_branch_review_evidence( }, "branch review evidence") if document["schema_version"] != 1 or type(document["schema_version"]) is not int: raise ContractError("branch review evidence schema_version is invalid") - if document["workflow"] != "branch-review": - raise ContractError("branch review evidence workflow is invalid") if not isinstance(snapshot, dict): raise ContractError("frozen workflow snapshot must be an object") task, workspace = snapshot.get("task"), snapshot.get("workspace") if not isinstance(task, dict) or not isinstance(workspace, dict): raise ContractError("frozen workflow snapshot is incomplete") + if (task.get("workflow") not in {"branch-review", "issue-delivery"} + or document["workflow"] != task["workflow"]): + raise ContractError("review evidence workflow is invalid") candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) for field, expected, validator in ( ("candidate_sha256", candidate_sha256, _sha256), @@ -975,7 +976,7 @@ def make_branch_review_evidence( candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) document = { "schema_version": 1, - "workflow": "branch-review", + "workflow": task["workflow"], "candidate_sha256": candidate_sha256, "base_oid": base_oid, "target_oid": target_oid, diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py index e490ffb..0928207 100644 --- a/test/core/test_delivery_workflow.py +++ b/test/core/test_delivery_workflow.py @@ -215,6 +215,14 @@ def implementation_fixture(delay: float = 0) -> dict[str, object]: "delay_seconds": delay, } + @staticmethod + def clean_review_fixture() -> dict[str, object]: + return { + "verdict": "clean", + "summary": "The exact frozen candidate satisfies the task.", + "findings": [], + } + def wait_for_candidate(self, service: Service, run_id: str) -> dict[str, object]: deadline = time.monotonic() + 15 while time.monotonic() < deadline: @@ -226,6 +234,17 @@ def wait_for_candidate(self, service: Service, run_id: str) -> dict[str, object] time.sleep(0.05) self.fail(f"delivery candidate did not become ready: {service.status(run_id)}") + def wait_for_handoff(self, service: Service, run_id: str) -> dict[str, object]: + deadline = time.monotonic() + 15 + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] == "awaiting_host": + return status + if status["state"] in {"failed", "cancelled"}: + self.fail(f"delivery review terminalized early: {status}") + time.sleep(0.05) + self.fail(f"delivery handoff did not become ready: {service.status(run_id)}") + def assert_source_unchanged(self) -> None: self.assertEqual(resolve_commit(self.repo, "HEAD"), self.baseline) self.assertEqual(self.git(self.repo, "status", "--porcelain"), self.source_status) @@ -379,6 +398,45 @@ def test_durable_implementer_publishes_candidate_artifacts_once(self): self.assertEqual(event_types.count("delivery.candidate_ready"), 1) self.assert_source_unchanged() + def test_exact_candidate_is_reviewed_checked_and_published_for_lead(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "reviewed-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + resumed = service.resume(started["run_id"]) + self.assertTrue(resumed["launched"]) + status = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], status["version"], "fixture-host", + ) + packet = claimed["handoff"]["packet"] + + store = Store(service.database, service.artifacts) + self.addCleanup(store.close) + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual([item["role"] for item in attempts], ["implementer", "reviewer"]) + self.assertEqual(packet["workflow"], "issue-delivery") + self.assertEqual( + packet["candidate_sha256"], snapshot["candidate"]["candidate_sha256"], + ) + self.assertTrue(packet["evaluation"]["accept_allowed"]) + self.assertEqual(packet["checks"][0]["status"], "passed") + self.assertEqual( + {item["name"] for item in packet["artifacts"]}, + { + f"review-{attempts[1]['id']}.json", + f"checks-{attempts[1]['id']}.json", + f"evaluation-{attempts[1]['id']}.json", + f"review-attempt-{attempts[1]['id']}.json", + }, + ) + self.assert_source_unchanged() + def test_live_implementer_cannot_be_resumed_into_a_second_writer(self): service = Service(self.runtime) started = service.start( diff --git a/test/core/test_review_workflow.py b/test/core/test_review_workflow.py index a8bce82..d916a9b 100644 --- a/test/core/test_review_workflow.py +++ b/test/core/test_review_workflow.py @@ -302,6 +302,28 @@ def test_combined_evidence_recomputes_gates_profile_and_accounting(self): with self.assertRaisesRegex(ContractError, message): validate_branch_review_evidence(invalid, self.snapshot) + def test_issue_delivery_review_uses_the_same_candidate_bound_contract(self): + task = copy.deepcopy(self.task) + task["workflow"] = "issue-delivery" + task["scope"]["write_paths"] = ["src/example.py"] + snapshot = copy.deepcopy(self.snapshot) + snapshot["task"] = task + + prompt = build_review_prompt(task, self.workspace) + evidence = make_branch_review_evidence( + snapshot, self.review, [self.check], + ) + + self.assertIn("read-only reviewer", prompt) + self.assertEqual(evidence["workflow"], "issue-delivery") + self.assertEqual( + validate_branch_review_evidence(evidence, snapshot), evidence, + ) + stale = copy.deepcopy(evidence) + stale["workflow"] = "branch-review" + with self.assertRaisesRegex(ContractError, "workflow is invalid"): + validate_branch_review_evidence(stale, snapshot) + if __name__ == "__main__": unittest.main() From 5298b15fb9f3b2eeb06beae991b28acd79bd1949 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sat, 26 Sep 2026 08:03:38 -0700 Subject: [PATCH 086/197] WIP checkpoint: evaluate Jev and Laya; preserve M5 review checkpoint (2026-09-26 08:03) --- ...02-surface-independent-engineering-team.md | 10 + .../engineering-team/DECISION-CLASSIFIERS.md | 198 ++++++++++++++++++ docs/plans/engineering-team/IMPLEMENTATION.md | 6 +- docs/plans/engineering-team/M5-STATUS.md | 21 +- docs/plans/engineering-team/RESUME.md | 34 ++- .../engineering-team/SELECTION-AND-COUNCIL.md | 11 + docs/plans/engineering-team/START-HERE.md | 14 +- docs/plans/engineering-team/backlog.json | 36 ++++ 8 files changed, 318 insertions(+), 12 deletions(-) create mode 100644 docs/plans/engineering-team/DECISION-CLASSIFIERS.md diff --git a/docs/adr/ADR-002-surface-independent-engineering-team.md b/docs/adr/ADR-002-surface-independent-engineering-team.md index ff447e1..74b15cc 100644 --- a/docs/adr/ADR-002-surface-independent-engineering-team.md +++ b/docs/adr/ADR-002-surface-independent-engineering-team.md @@ -117,6 +117,16 @@ Routing starts with a small versioned preference list per role/task class. Filte Selection is automatic by default. A user may pin a validated profile for any role; unpinned roles remain automatic. The selected profile defines the permitted toolbox, and the worker selects actual tool calls within it. Overrides have explicit fallback behavior. Stable role aliases resolve to qualified concrete profiles and are frozen per run. Discovery/evaluation can propose replacements; a reviewed update policy may authorize guarded automatic binding promotions, while policy changes remain reviewed. See the [selection amendment](../plans/engineering-team/SELECTION-AND-COUNCIL.md) and [model lifecycle](../plans/engineering-team/MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md). +**September 26 amendment:** Evaluate optional typed decision helpers under +[DECISION-CLASSIFIERS.md](../plans/engineering-team/DECISION-CLASSIFIERS.md). +Default-off shadow experiments may test semantic task/profile, skill and +context suggestions. A separately reviewed policy may consume frozen hints +after held-out validation; exact eligibility, pins, permissions, billing, +capacity and lead authority stay unchanged. This narrowly extends the original +deferral of learned routing; it authorizes neither autonomous training/policy +changes nor a required hosted service. The current no-model route remains the +baseline and fallback, and M1–M5 acceptance gates are unchanged. + Account pools span applications and repositories where the underlying allowance is shared. Provider observations have sources and expiry times; unknown allowance is unknown. DevSquad's concurrency reservations do not reserve quota with a provider. Spend estimates, token counts, characters and subscription allowance are distinct measurements. ## Learning and documentation are part of completion diff --git a/docs/plans/engineering-team/DECISION-CLASSIFIERS.md b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md new file mode 100644 index 0000000..bab7b0f --- /dev/null +++ b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md @@ -0,0 +1,198 @@ +# Typed decision classifiers: Jev, Laya and DevSquad + +Decision date: **2026-09-26**. Status: **source evaluation and planning +amendment; runtime integration and performance evaluation are pending**. + +## Recommendation + +Evaluate a small typed classifier as an optional **decision helper**, not a +replacement lead, coding worker or authority boundary. Start with **Laya in a +local, default-off shadow experiment**; keep **Jev as an optional hosted +comparator** requiring explicit API spending and data-sharing authorization. +Existing paid coding subscriptions do not establish access to the Jev API. +Do not add a mandatory model dependency or delay M5's delivery/revision loop. + +The current [router](../../../plugin/core/src/devsquad/router.py) is deterministic +and makes no model call. A classifier adds routing latency. Its potential value +is better task/profile matching, fewer unnecessary tool loads, less irrelevant +context and less downstream rework. No DevSquad speedup, allowance saving or +number of saved subscription windows has been measured. + +This evaluation inspected public documentation, benchmark reports and pinned +Laya source. It did **not** install weights, run inference, send project data to +TypeSafe, or benchmark either model on this Mac or DevSquad tasks. + +## What was verified + +| Candidate | Relevant capability | Constraint for this project | +|---|---|---| +| **Jev, TypeSafe System One** | Hosted typed choice, rubric-score and boolean decisions; multiple questions over shared state | Not a code generator. A separate network/API dependency with separate billing and privacy decisions | +| **Laya** | Apache-2.0 local typed decision models with English, multilingual and typed-workflow variants | Optional heavy inference environment and model assets; local latency, memory, calibration and engineering-task quality are unverified | + +Jev's documented current version is `jev-1.13.0`; `jev-latest` is mutable. +Published pricing is $0.042 per million input tokens, with output tokens free. +The documented limits are 64k aggregate context and 32k for state plus the +longest question. These are provider claims/current specifications, not local +measurements. Pin the concrete model and recheck terms at integration time. +[Jev models and pricing](https://docs.typesafe.ai/models). + +Choice/score confidence describes the output distribution; it is not proof +that a decision is correct. Jev documents weaknesses in exact numeric tasks, +indirection and adversarial input. Its coding-agent guidance explicitly +distinguishes typed decisions from generation and tool execution. +[Confidence](https://docs.typesafe.ai/confidence), +[known weaknesses](https://docs.typesafe.ai/model-jaggedness/jev-1.13), +[coding-agent boundary](https://docs.typesafe.ai/introduction/coding-agents). + +The inspected Laya source is package `0.3.20`, commit +`4066d5d5fbf08b66c6757ddeedbd797bd7655bc0`. Its dependencies include PyTorch and +Transformers, unlike DevSquad's dependency-free core. Revision/hash pinning is +supported but must be selected explicitly. Its `Router` chooses among Laya +checkpoints using language/script and optional question-ID heuristics; it is +**not a coding-provider selector**. The bundled Hub revision observed was +`55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851`; pin individual assets in the eventual +experiment manifest, not just a moving model name. +[Package](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/pyproject.toml), +[router](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/laya/router.py), +[revision handling](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/laya/revisions.py), +[model](https://huggingface.co/convaiinnovations/laya). + +Laya's published 32.8 ms T4 one-question median is not Mac end-to-end latency. +Its Jev comparison uses third-party numbers, not a matched head-to-head run. +The reported model-routing task measures domain classification, not which +coding model delivers a correct patch. Results also show language/calibration +weaknesses and degradation with many labels. Its own guidance suggests small +option sets. These findings justify evaluation, not automatic adoption. +[Pinned benchmark report](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/BENCHMARKS.md). + +Laya can truncate state/question content; long-input aggregation retains a +window's confidence, not a calibrated document-level probability. Choice/score +entropy confidence and maximum answer probability are different fields. Keep +raw scores and our separately evaluated calibration, rather than normalizing +every provider's “confidence” into a supposed universal success probability. +[Pinned inference code](https://github.com/NandhaKishorM/laya/blob/4066d5d5fbf08b66c6757ddeedbd797bd7655bc0/laya/agent.py). + +## Other places to use this pattern + +Priority is an evaluation order, not a promise to enable every use case. +P1/P2/P3 below indicate sequence, not defect severity. + +| Priority / use | Bounded classifier output | Potential benefit and required boundary | +|---|---|---| +| P1 — Task intake and routing hints | Task family, complexity band, ambiguity/missing-information flags; rank already-qualified profiles | Improve template/profile matching. The lead validates the task; hints cannot lower its quality floor, change scope or invent capabilities | +| P1 — Skill and tool shortlist | Up to a few relevant IDs, including `none`/`uncertain`, from an approved catalog | Avoid irrelevant tool/skill loading. Read selected instructions fully; retain catalog access, validate actual arguments and keep worker permissions unchanged | +| P1 — Context and retrieval ranking | Relevance labels over bounded file, diff, log or retrieved-passage candidates | Reduce irrelevant context. Never remove mandatory instructions, acceptance criteria, failed checks or review dissent; retain source references and measure evidence recall | +| P2 — Failure triage | Semantic category for otherwise unclassified diagnostics, plus suggested next action | Group unknown failures for investigation. Exact auth/rate/timeout codes, exit status and protocol evidence remain authoritative; no automatic retry or paid fallback | +| P2 — Review and test attention | Finding clusters, subsystem/risk tags, relevant optional test IDs | Focus a reviewer and suggest extra checks. Preserve original findings and provenance; never suppress a blocker, waive independent review or skip required tests | +| P2 — Outcome learning and documentation | Proposed repair/failure labels, affected requirement/doc IDs | Make comparable cohorts and identify documentation gaps. Labels need evidence and correction; they cannot declare acceptance, author documentation or promote defaults | +| P3 — Council/escalation suggestions | Disagreement/ambiguity flags linked to evidence | Help the existing lead spot cases worth deliberating. No classifier-only council trigger, extra worker launch or change to C1's explicit budget and evaluation gate | + +Skill suggestion has unusually relevant prior evidence: TypeSafe reports fewer +wrong and unnecessary skill loads in a 488-request, 182-skill experiment. It +used Jev 1.12 and synthetic requests, not DevSquad; the agent retained access to +the full skill index. Treat this as a testable hypothesis, not a reproduced gain. +[Official skill-suggestion experiment](https://docs.typesafe.ai/cookbooks/skill_suggestion). + +The target is catalogs passed to DevSquad-managed workers. This does not allow +DevSquad to bypass the host's skill instructions, alter global AI settings, or +replace provider-native internal tool selection it cannot control. Context +selection is ranking, not generative summarization; complicated synthesis and +patch writing remain lead/worker tasks. + +Do **not** use a probabilistic classifier for quota arithmetic/reset times, +authentication truth, observed model identity, capability verification, lock +ownership, crash recovery, Git/candidate integrity, permission grants, mandatory +check acceptance, or the sole security/redaction gate. Existing exact checks +are cheaper and authoritative for these jobs. + +## Minimal architecture amendment + +This extends M6 evaluation and M7 optional packaging. It does not reopen M1–M5 +or remove any existing acceptance gate. Automatic self-training and unreviewed +learned policy changes remain deferred. + +1. **Three explicit modes:** `off` (default, current behavior), `shadow` + (save suggestions without changing execution), and `advisory` (only a + reviewed, versioned policy after the use-case gate passes). Classification + is optional preprocessing; the router consumes frozen validated data and + remains deterministic. A cache miss never silently enables inference. +2. **Constrain before ranking:** exact policy filters establish permissions, + billing, capabilities, verified identity, quality and current capacity. + Only permitted candidates may be scored. Honor pins and explicit fallback; + revalidate volatile capacity at reservation. Suggestions cannot change task + class/requirements or expand the eligible set. No eligible candidate still + means blocked, not “let the classifier choose.” +3. **Typed adapter contract:** versioned purpose/question/rubric, input and + evidence hashes, candidate IDs, model/runtime revision, language and + truncation metadata; output labels, raw probability vector, provider-specific + confidence, optional separately versioned calibration and an abstain reason. + Validate schema, known IDs, finite/ranged scores and distributions before + use. Invalid, unsupported, truncated, late or uncertain output falls back + to the unchanged deterministic behavior and is recorded. +4. **Freeze and replay:** save observations with the run/experiment. Cache keys + cover scope/evidence, candidate and catalog hashes, question/options/rubric, + schema, model/runtime, calibration and language. Changed inputs invalidate + reuse; resume reuses the saved observation instead of paying again. Keep + raw private content outside tracked evidence. Predictions are untrusted + data and cannot supply shell commands, policy text or new permissions. + An interrupted external call with an unknown outcome is not proof of no + charge: record it as indeterminate and abstain rather than automatically + resubmitting on resume. Any explicit retry needs its own budget reservation. +5. **Bound the extra work:** explicitly budget calls, input size, wall time, + local memory and any authorized API cost. Account for classifier calls, + worker launches and native usage separately. Batch related questions only + within budget; use bounded retries and cancellation. A timeout must not + outlive its run or consume the worker's entire remaining deadline. +6. **Keep installation optional:** no heavy imports in core CLI or hooks; no + hook network calls, inference or model downloads. Use an isolated optional + environment and pinned assets. An owned bounded process can reuse a loaded + model within a run; no permanent service/extra MCP server is required. Model + downloads are explicit setup, never an automatic fallback. Missing extras, + unsupported hardware or a failed helper must preserve ordinary operation. + +Jev experiments additionally require an approved endpoint/model, allowed data +classes, redaction policy, spending ceiling and explicit retry policy. Its API +supports typed responses and reports usage; record actual returned usage, +errors and concrete model identity rather than inferring usage from a request. +[Official API](https://docs.typesafe.ai/api). + +## Spec-based iterations and acceptance + +These are small work packages within M6's existing experiments/learning work, +not new prerequisite milestones. M4's external live-host blocker stays open; +independent offline evaluation can proceed once the M5 evidence shape is stable. +M5 disposition, bounded revisions and receipts remain the immediate next work. + +| Package | Deliverable | Required proof before advancing | +|---|---|---| +| **M6-D1 — Contract and baseline** | Strict optional decision schema, fake adapter, run/cache accounting, redacted labeled corpus and frozen experiment spec | Default-off equivalence; malformed/unknown/NaN output, candidate/pin/quality/permission attacks, input drift, cancellation and crash/resume tests. Zero unauthorized selections or duplicated paid calls on replay | +| **M6-D2 — Local shadow trial** | Pinned optional Laya adapter; start with task/profile hints and skill shortlist; context ranking next only if justified | Paired held-out comparison with static/heuristic and existing-lead baselines; cold/warm timing and memory on the actual Mac; explicit language/long-input abstention; no impact on ordinary runs | +| **M6-D3 — Adoption decision** | Evidence-backed keep-off or narrowly scoped advisory policy, rollback receipt, optional separately authorized Jev comparison | Use-case-specific quality/cost gate passes before opt-in; failed/inconclusive experiments remain off. Re-run held-out and policy-boundary tests on any model/rubric/calibration change | + +Freeze the dataset split, metrics, thresholds, sample-size rationale and resource +ceilings **before** evaluating the held-out set. Split by issue/repository family +to prevent near-duplicate leakage. Include real engineering tasks, no-match +cases, long/noisy inputs, English/Hinglish/other supported languages, adversarial +instructions and quota-constrained candidate sets. Synthetic fixtures test +mechanics; they do not establish user-facing quality. + +Measure task-family/shortlist accuracy, abstention coverage and calibration, +context evidence recall, inappropriate downgrade rate, actual check/review +outcomes, escaped defects, lead repairs, total latency, peak memory and observed +worker/token usage. The cheapest sufficient profile is established from verified +outcomes, not a model-brand label or another classifier's answer. Report sample +sizes, uncertainty intervals and unavailable usage; do not convert worker counts +or tokens into fixed Plus-window consumption. + +Adoption requires all authority/integrity tests passing, no observed critical +evidence loss or constraint violation, quality meeting the predeclared +non-inferiority margin, and a material end-to-end benefit after helper overhead. +Use a predeclared target (initial proposal: at least 10% less observed scarce +worker usage **or** lead rework time) with adequate uncertainty bounds; a tiny +pilot cannot prove the production gate. Scarce usage must be observed, not an +invented quota conversion. Limited or inconclusive data means keep shadow/off. +Jev and Laya need the same inputs/options and end-to-end accounting to support +any head-to-head claim. + +Expand to P2/P3 only through separate small specs and measurements. A successful +skill selector does not establish a safe model router, failure handler or judge. diff --git a/docs/plans/engineering-team/IMPLEMENTATION.md b/docs/plans/engineering-team/IMPLEMENTATION.md index bcd3fee..624182d 100644 --- a/docs/plans/engineering-team/IMPLEMENTATION.md +++ b/docs/plans/engineering-team/IMPLEMENTATION.md @@ -119,9 +119,11 @@ Work in this order: 6. Implement the [model lifecycle](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md): budgeted qualification, limited trials, reviewed or explicitly enabled guarded automatic binding promotion, compare-and-swap binding versions, last-qualified fallback and rollback. Promotions stay within allowed templates, quality criteria, permissions and billing authority. They write local decision receipts and affect new runs only. The default remains reviewed until evaluation gates and policy enable guarded automation. +7. Execute the optional [typed decision-classifier evaluation](DECISION-CLASSIFIERS.md), packages M6-D1–D3, within the existing experiment budget. Begin with strict fake-adapter contracts and a pinned local Laya shadow trial for task/profile hints and skill shortlists; consider context ranking next. Jev is a hosted comparator only after explicit API spending/data authorization. Compare with the existing no-model router and lead on held-out engineering outcomes, overhead and rework. Keep-off/inconclusive is a valid adoption decision, not a reason to relax the gate. + **Acceptance gate:** Two projects sharing one pool obey a fresh exhausted weekly window despite available short-window capacity. Stale/unknown values never become zero; external usage changes do not get assigned to one worker. Concurrency reservations release only after ownership is reconciled. A paid API fallback is excluded unless allowed. A failed original attempt later repaired by another model produces final success without crediting the original as independently successful. A late escaped bug updates outcome history. A one-variable fixture experiment produces a traceable no-change or promotion proposal, with all failures and a rollback version; insufficient evidence leaves active policy unchanged. Re-run a held-out fixture after policy change and exercise rollback. -**Boundary:** No automatic learned router. Experiment budgets default off; activate only through explicit versioned policy. +**Boundary:** No autonomous learned-policy changes or model training. Experiment budgets and decision helpers default off; activate only through explicit versioned policy. The September 26 [classifier amendment](DECISION-CLASSIFIERS.md) permits frozen semantic suggestions and, after its use-case gate, reviewed opt-in ranking within an already-qualified set. The deterministic router, user pins, quality floors, capacity/permission/billing checks and lead acceptance remain authoritative. No classifier is required for normal operation. **Release gate:** A new model cannot become default from discovery alone. Insufficient evidence, unsupported effort or widened permissions/billing prevents automatic promotion. Changed metadata revalidates only affected profiles; same-ID backing changes retain unknown revision when unobservable. An approved binding update affects a newly started run while an existing run and concrete override remain pinned. Concurrent promotions conflict on stale binding versions. A regression reverts to an available qualified binding and records why. All qualification/trial launches share the configured experiment and account-pool budgets. @@ -137,6 +139,8 @@ Work in this order: 4. Add offline CI for legacy and core suites (Bash 3.2/macOS compatibility and chosen Python floor/current version), package-content checks, native protocol compatibility fixtures and optional MCP tests. Live provider/app tests remain explicit bounded smoke runs. Document supported protocol ranges, capability drift and the optional native Claude→Codex session-import path, while retaining portable artifact handoffs for every host. 5. Verify terminal CLI, Codex App, Claude Code App local Code tab, Antigravity local IDE/CLI and Grok Build against the same saved runtime. Check native capabilities/profile identity as used, not by brand inference. +6. If the classifier trial is retained, package it as an isolated optional extra with pinned model assets and explicit setup. Test absent extras, offline operation, unsupported hardware, cancellation and default-off equivalence. Hooks remain inference/download-free and network-free; neither a hosted key nor heavy model dependencies become a core-install requirement. + **Acceptance gate:** Fresh standalone install works without Claude installed; existing Claude plugin install contains `plugin/core` contents correctly. Reinstall creates no duplicate hook/server registration and a release update does not break a running job. Record actual start/observe/handoff-or-cancel receipts from every listed surface. Complete one end-to-end delivery with a different-model reviewer after installation. Documentation commands run as written. If an installed host cannot support an operation, retain that item as blocked with exact evidence instead of declaring universal support. ## Optional C1 — Selective Council decisions diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index ebfb544..1af7be9 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -11,7 +11,7 @@ two-harness gate passes. | Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | verified offline at `0e88d73` | | Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `0e88d73` | | Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | router verified; workflow pending | -| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | pending | +| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | first-candidate review/check/handoff verified offline at `30b98df`; revision/stale-evidence workflow gate pending | | Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | pending | | Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | pending | | Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | pending | @@ -71,3 +71,22 @@ uses the non-zombie process-group inventory before declaring ambiguity and has a direct regression. The isolated service/supervisor/delivery suites and the complete discovery pass after that fix. Plan 07-02 now owns independent review, checks, disposition and revision behavior. + +## Plan 07-02 checkpoint 1 + +At `30b98df`, review prompts/evidence accept `issue-delivery` only when the +workflow matches the frozen task. A pending offline review fixture is bound +to the actual candidate after implementation; the durable reviewer then runs +read-only, imports its candidate-bound evidence, executes trusted checks in +the separate worktree, and publishes the correct delivery lead handoff. + +The end-to-end offline regression proves implementation → explicit candidate +resume → review/checks → lead claim, with matching candidate hashes and no +source-checkout mutation. This is not native identity verification or a live +two-harness result. Lead completion, revise-to-implementer iterations, full +terminal history and the remaining M5 fault/live gates are still pending. + +The complete core discovery is **226 tests, suite OK with 2 optional-SDK +skips**, with `ResourceWarning` promoted to error; **220 Bash assertions** +passed. The next implementation slice is Plan 07-02 Task 3, retaining Task 2's +unproven live/stale-revision requirements rather than marking M5 complete. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index af3a7c0..826c2b3 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,7 +2,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 23, 2026 +## Current position — September 26, 2026 - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` remains the published runtime @@ -102,6 +102,22 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne out-of-scope edits fail without a candidate. The complete gate is 224 core tests and 220 Bash assertions. A reproduced macOS zombie-only process-group ambiguity was fixed with a non-zombie inventory check and direct regression. +- M5 Plan 07-02 first-candidate review/check/handoff is verified offline at + `30b98df`. The durable implementer creates a frozen candidate, explicit + resume launches read-only review, separate trusted checks validate that + candidate, and the host receives an `issue-delivery` handoff bound to its + hash. The complete gate is 226 core tests discovered (suite OK, 2 optional + SDK skips) and 220 Bash assertions. This does not complete lead disposition, + revise-to-implementation, all terminal reports or the live two-harness gate. +- The user's Jev/Laya request is evaluated in + [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan + amendment adds M6-D1–D3: default-off contracts/baseline, local Laya shadow + trial, and measured keep-off or reviewed adoption. It prioritizes routing + hints, skill/tool shortlists and context ranking, followed by failure triage, + review attention and outcome labels. No weights/inference/API spending or + runtime routing changes occurred. Jev comparison needs separate API/data + authorization; classifier suggestions never become permission/acceptance + authority. M5 remains the immediate implementation priority. ## Completed and preserved @@ -113,7 +129,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 224 tests passed through completed M5 Plan 07-01 with warnings promoted to errors | +| Python core discovery | 226 discovered at `30b98df`; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -142,7 +158,7 @@ the earlier apparent nonresponses. The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md), [M2-STATUS.md](M2-STATUS.md) and [M3-STATUS.md](M3-STATUS.md). [backlog.json](backlog.json) marks M1–M3 complete, M4 blocked on its external -Claude live gate, and M5 next. +Claude live gate, and M5 in progress. M6 classifier work packages remain pending. Unauthenticated, unsupported or permission-blocked provider paths are not advertised as verified. @@ -150,12 +166,18 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Execute M5 Plan 07-02: generalize candidate-bound review/check evidence for - `issue-delivery`, publish the lead handoff, and make a bounded `revise` - disposition return to the fenced implementer with all iterations retained. +2. Continue M5 Plan 07-02 Task 3 from `30b98df`: complete delivery lead + disposition and make bounded `revise` return to the fenced implementer; + retain every candidate/attempt/check/disposition in terminal reports. + Exercise seeded repair, stale-candidate rejection, mandatory-check blocking, + fallback/deadline bounds and crash recovery. First-candidate offline + review/check/handoff already works; do not rebuild that slice. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. +4. When M5's evidence shape is stable, execute the small M6 decision-helper + work packages alongside other independently ready M6 work. Keep experiments + off by default and preserve all existing M4/M5/live acceptance gates. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. diff --git a/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md index 52ba697..f8db0ff 100644 --- a/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md +++ b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md @@ -11,6 +11,7 @@ | Discover installed harnesses, models, supported efforts and tools | DevSquad's discovery/probes | | Connect accounts, identify shared allowance pools, set spending/access limits | User setup, assisted by discovery | | Frame the task, scope, acceptance criteria and required capabilities | Current host lead, or supplied terminal task | +| Suggest task labels, relevant skills/context or eligible-profile rankings | Optional evaluated decision helper; advisory data, never authority | | Select model, effort and permitted toolbox for each role | Deterministic router applying versioned policy and current availability | | Decide which permitted tool to call during work | Selected worker, inside the assigned permissions | | Pin a particular configuration for this task | User override, resolved by the host into a validated profile | @@ -63,6 +64,16 @@ Reports explain selections, excluded alternatives, explicit overrides, escalatio The [model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) adds stable aliases, automatic discovery and bounded qualification. After calibration, an enabled `guarded_auto` policy can promote tested bindings without per-release manual edits. The reviewed policy remains the authority; discovery or council votes alone cannot promote a candidate. +The September 26 [Jev/Laya decision-helper amendment](DECISION-CLASSIFIERS.md) +adds a default-off, local-first evaluation within M6. Frozen suggestions may +inform only a reviewed policy after use-case-specific held-out validation. +The router remains deterministic: validate requirements, pins and eligible +profiles before considering suggestions; invalid/uncertain/missing output uses +the existing policy. Helpers cannot lower task quality requirements, relax +capacity, grant tools or trigger extra workers. Skill/context recommendations +retain mandatory instructions and evidence. This is not a second planner or a +change to C1's independent evaluation and explicit-invocation gates. + ## What LLM Council actually contributes Studied **Karpathy's original repository**, pinned at [`92e1fcc`](https://github.com/karpathy/llm-council/tree/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131). This is a source review, not a performance benchmark or a survey of forks. Its author describes it as an exploratory, unsupported project. [Original README](https://github.com/karpathy/llm-council/blob/92e1fccb1bdcf1bab7221aa9ed90f9dc72529131/README.md). diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index a04cb7c..92b66c1 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -1,9 +1,10 @@ # DevSquad: coding-agent entry point **Build status: implementation in progress.** After an interruption, read -[RESUME.md](RESUME.md) first and compare it with current Git state. M1 has code -and passing offline evidence; its remaining gates are recorded in -[M1-STATUS.md](M1-STATUS.md). Do not restart the architecture exercise. +[RESUME.md](RESUME.md) first and compare it with current Git state. M1–M3 are +accepted; M4's actual Claude handoff is externally blocked and M5 is in +progress. See [backlog.json](backlog.json) for evidence. Do not restart the +architecture exercise. **Full-build assignment:** Use [SOL-HANDOFF.md](SOL-HANDOFF.md) for the user's request to have Sol execute everything, test thoroughly and make normal use simple. It includes M1–M7 plus the opt-in Council feature, and adds guided task entry over the same contracts. @@ -33,6 +34,11 @@ Selection is automatic by default, with validated per-role profile overrides. Re Also read the [native adapters and model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md): use verified Codex app-server capabilities, stable profile aliases, automatic catalog updates and qualified binding promotions. These refine M1/M3/M6/M7; they add no prerequisite milestone and do not require rewriting workflows for each model release. +The September 26 [Jev/Laya evaluation and decision-helper amendment](DECISION-CLASSIFIERS.md) +adds optional M6 experiments for routing hints, skill selection and context +ranking, with further bounded uses prioritized. It changes no runtime defaults +and does not delay M5 or authorize paid API usage. + ## Copyable execution brief ```text @@ -70,7 +76,7 @@ Keep planned and implemented features visibly separate. Do not mark M4/M7 comple First usable product: **a saved branch review**. Next: **one bounded code change reviewed by another model**. Two functioning harnesses are sufficient to prove the engineering workflow; M7 verifies access from every requested local surface. Do not force every provider into every run. -Defer a dashboard, universal DAG builder, remote execution service, automatic model training/router, plugin marketplace, autonomous merges and scheduled documentation jobs. Existing plugin behavior remains available while the new runner is opt-in; switching hook suggestions to the new route source happens only after its gate passes. +Defer a dashboard, universal DAG builder, remote execution service, automatic model training or unreviewed learned policy changes, plugin marketplace, autonomous merges and scheduled documentation jobs. Optional evaluated decision hints are bounded by the classifier amendment, not a replacement for deterministic policy. Existing plugin behavior remains available while the new runner is opt-in; switching hook suggestions to the new route source happens only after its gate passes. This packet began as architecture only. Current implementation and live-probe evidence are tracked in RESUME.md, the milestone status and backlog; they do diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 4a95c0d..7555680 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -8,6 +8,7 @@ "implementation": "IMPLEMENTATION.md", "selection_and_council": "SELECTION-AND-COUNCIL.md", "model_lifecycle_and_native_adapters": "MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md", + "decision_classifiers": "DECISION-CLASSIFIERS.md", "execution_brief": "SOL-HANDOFF.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", @@ -239,6 +240,15 @@ "artifact": "M5-STATUS.md", "recorded_at": "2026-09-23T18:37:00+05:30", "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "30b98df", + "command_or_action": "226 core tests discovered with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions and exact-candidate delivery/review regressions", + "outcome": "Offline delivery now advances from a fenced local candidate through read-only review and separate trusted checks to a candidate-bound issue-delivery lead handoff; disposition, bounded revisions, complete receipts and live different-model/two-harness proof remain open", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-26T15:01:21Z", + "availability": "tracked_tests" } ], "blocker": null @@ -249,6 +259,32 @@ "depends_on": ["M4", "M5"], "status": "pending", "acceptance_section": "M6 — Make capacity and improvement evidence useful", + "decision_helper_work_packages": [ + { + "id": "M6-D1", + "title": "Optional typed decision contract and frozen evaluation baseline", + "status": "pending", + "specification": "DECISION-CLASSIFIERS.md", + "evidence": [] + }, + { + "id": "M6-D2", + "title": "Pinned local Laya shadow trial for task/profile and skill hints", + "depends_on": ["M6-D1"], + "status": "pending", + "specification": "DECISION-CLASSIFIERS.md", + "evidence": [] + }, + { + "id": "M6-D3", + "title": "Measured keep-off or reviewed advisory adoption decision", + "depends_on": ["M6-D2"], + "status": "pending", + "specification": "DECISION-CLASSIFIERS.md", + "evidence": [], + "hosted_comparison": "Optional Jev evaluation requires separate explicit API spending and data-sharing authorization; local evaluation does not depend on it" + } + ], "evidence": [], "blocker": null }, From 70e59cb163af3c441e762d74aa161b376553d063 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sat, 26 Sep 2026 10:26:30 -0700 Subject: [PATCH 087/197] WIP checkpoint: prepare capped Jev routing pilot (2026-09-26 10:26) --- .../engineering-team/DECISION-CLASSIFIERS.md | 50 ++- docs/plans/engineering-team/IMPLEMENTATION.md | 2 +- docs/plans/engineering-team/RESUME.md | 20 +- .../engineering-team/SELECTION-AND-COUNCIL.md | 4 +- docs/plans/engineering-team/START-HERE.md | 4 +- docs/plans/engineering-team/backlog.json | 16 +- .../experiments/jev-pilot-v1.json | 130 +++++++ test/core/probes/jev_decision_eval.py | 364 ++++++++++++++++++ test/core/test_jev_decision_probe.py | 128 ++++++ 9 files changed, 692 insertions(+), 26 deletions(-) create mode 100644 docs/plans/engineering-team/experiments/jev-pilot-v1.json create mode 100644 test/core/probes/jev_decision_eval.py create mode 100644 test/core/test_jev_decision_probe.py diff --git a/docs/plans/engineering-team/DECISION-CLASSIFIERS.md b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md index bab7b0f..3f0628d 100644 --- a/docs/plans/engineering-team/DECISION-CLASSIFIERS.md +++ b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md @@ -6,11 +6,14 @@ amendment; runtime integration and performance evaluation are pending**. ## Recommendation Evaluate a small typed classifier as an optional **decision helper**, not a -replacement lead, coding worker or authority boundary. Start with **Laya in a -local, default-off shadow experiment**; keep **Jev as an optional hosted -comparator** requiring explicit API spending and data-sharing authorization. -Existing paid coding subscriptions do not establish access to the Jev API. -Do not add a mandatory model dependency or delay M5's delivery/revision loop. +replacement lead, coding worker or authority boundary. Per the user's September +26 direction, start with a **single capped Jev synthetic pilot** because it +avoids Laya's local model setup. The pilot is limited to one billable request, +no retries, a $0.01 ceiling and no repository/private task content. Move to a +local, default-off Laya trial if Jev becomes materially costlier, cannot be +accessed, or fails the quality/latency gate. Existing paid coding subscriptions +do not establish access to the Jev API. Do not add a mandatory model dependency +or delay M5's delivery/revision loop after the bounded probe. The current [router](../../../plugin/core/src/devsquad/router.py) is deterministic and makes no model call. A classifier adds routing latency. Its potential value @@ -18,9 +21,12 @@ is better task/profile matching, fewer unnecessary tool loads, less irrelevant context and less downstream rework. No DevSquad speedup, allowance saving or number of saved subscription windows has been measured. -This evaluation inspected public documentation, benchmark reports and pinned -Laya source. It did **not** install weights, run inference, send project data to -TypeSafe, or benchmark either model on this Mac or DevSquad tasks. +The initial evaluation inspected public documentation, benchmark reports and +pinned Laya source. It did **not** install weights or run Laya inference. The +tracked [Jev pilot specification](experiments/jev-pilot-v1.json) and validated +[probe](../../../test/core/probes/jev_decision_eval.py) are now ready, but no +TypeSafe request has run because this environment has no `TYPESAFE_API_KEY` and +the console is at its login screen. The live result remains pending. ## What was verified @@ -156,6 +162,30 @@ supports typed responses and reports usage; record actual returned usage, errors and concrete model identity rather than inferring usage from a request. [Official API](https://docs.typesafe.ai/api). +### Immediate capped Jev pilot + +The v1 pilot batches eight synthetic task cases and 24 task-family, +execution-tier and specialist-skill choices into **one** Jev 1.13 request. The +dry run is 19,219 request bytes. At the frozen published price, even the model's +documented 64k aggregate context limit would cost about $0.002688, below the +$0.01 ceiling. This calculation is only a preflight cap; the receipt must use +the API's returned `input_tokens`. The probe has no retry path, never accepts a +key on the command line and emits only case IDs, choices, probability vectors, +latency and usage—not task text. + +```bash +python3 test/core/probes/jev_decision_eval.py +# Export TYPESAFE_API_KEY without placing its value in shell history, then run: +python3 test/core/probes/jev_decision_eval.py --execute \ + --output "$HOME/.devsquad/private-probes/jev-pilot-v1.json" +``` + +Do not copy the key or private receipt into Git. If the provider's current price +makes one maximum-context request exceed $0.01, if returned usage breaches the +cap, or if access requires purchasing a larger commitment, stop without retry +and start the pinned Laya local trial. A smoke result only answers whether Jev +can follow this schema on synthetic cases; it cannot enable runtime routing. + ## Spec-based iterations and acceptance These are small work packages within M6's existing experiments/learning work, @@ -166,8 +196,8 @@ M5 disposition, bounded revisions and receipts remain the immediate next work. | Package | Deliverable | Required proof before advancing | |---|---|---| | **M6-D1 — Contract and baseline** | Strict optional decision schema, fake adapter, run/cache accounting, redacted labeled corpus and frozen experiment spec | Default-off equivalence; malformed/unknown/NaN output, candidate/pin/quality/permission attacks, input drift, cancellation and crash/resume tests. Zero unauthorized selections or duplicated paid calls on replay | -| **M6-D2 — Local shadow trial** | Pinned optional Laya adapter; start with task/profile hints and skill shortlist; context ranking next only if justified | Paired held-out comparison with static/heuristic and existing-lead baselines; cold/warm timing and memory on the actual Mac; explicit language/long-input abstention; no impact on ordinary runs | -| **M6-D3 — Adoption decision** | Evidence-backed keep-off or narrowly scoped advisory policy, rollback receipt, optional separately authorized Jev comparison | Use-case-specific quality/cost gate passes before opt-in; failed/inconclusive experiments remain off. Re-run held-out and policy-boundary tests on any model/rubric/calibration change | +| **M6-D2 — Capped Jev pilot** | One-request synthetic smoke, then a larger shadow comparison only if separately budgeted and justified | Exact model/usage/latency/cost receipt; strict response validation; no retry, task disclosure or runtime effect. Missing access, cost breach or poor results select the Laya fallback rather than weakening the gate | +| **M6-D3 — Local fallback and adoption decision** | Pinned optional Laya trial when triggered, followed by evidence-backed keep-off or narrowly scoped advisory policy and rollback receipt | Same cases and end-to-end accounting for any Jev/Laya comparison; use-case quality/cost gate before opt-in. Failed/inconclusive experiments remain off; model/rubric/calibration changes rerun held-out and boundary tests | Freeze the dataset split, metrics, thresholds, sample-size rationale and resource ceilings **before** evaluating the held-out set. Split by issue/repository family diff --git a/docs/plans/engineering-team/IMPLEMENTATION.md b/docs/plans/engineering-team/IMPLEMENTATION.md index 624182d..bd6d10a 100644 --- a/docs/plans/engineering-team/IMPLEMENTATION.md +++ b/docs/plans/engineering-team/IMPLEMENTATION.md @@ -119,7 +119,7 @@ Work in this order: 6. Implement the [model lifecycle](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md): budgeted qualification, limited trials, reviewed or explicitly enabled guarded automatic binding promotion, compare-and-swap binding versions, last-qualified fallback and rollback. Promotions stay within allowed templates, quality criteria, permissions and billing authority. They write local decision receipts and affect new runs only. The default remains reviewed until evaluation gates and policy enable guarded automation. -7. Execute the optional [typed decision-classifier evaluation](DECISION-CLASSIFIERS.md), packages M6-D1–D3, within the existing experiment budget. Begin with strict fake-adapter contracts and a pinned local Laya shadow trial for task/profile hints and skill shortlists; consider context ranking next. Jev is a hosted comparator only after explicit API spending/data authorization. Compare with the existing no-model router and lead on held-out engineering outcomes, overhead and rework. Keep-off/inconclusive is a valid adoption decision, not a reason to relax the gate. +7. Execute the optional [typed decision-classifier evaluation](DECISION-CLASSIFIERS.md), packages M6-D1–D3, within the existing experiment budget. Begin with strict fake-adapter contracts and the user-authorized one-request, $0.01 Jev synthetic pilot for task/profile hints and skill shortlists. Use no private task content and no retry. Move to a pinned local Laya shadow trial if hosted access/cost or measured quality warrants its setup; consider context ranking next. Compare with the existing no-model router and lead on held-out engineering outcomes, overhead and rework. Keep-off/inconclusive is a valid adoption decision, not a reason to relax the gate. **Acceptance gate:** Two projects sharing one pool obey a fresh exhausted weekly window despite available short-window capacity. Stale/unknown values never become zero; external usage changes do not get assigned to one worker. Concurrency reservations release only after ownership is reconciled. A paid API fallback is excluded unless allowed. A failed original attempt later repaired by another model produces final success without crediting the original as independently successful. A late escaped bug updates outcome history. A one-variable fixture experiment produces a traceable no-change or promotion proposal, with all failures and a rollback version; insufficient evidence leaves active policy unchanged. Re-run a held-out fixture after policy change and exercise rollback. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 826c2b3..4acd8aa 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -111,13 +111,18 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne revise-to-implementation, all terminal reports or the live two-harness gate. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan - amendment adds M6-D1–D3: default-off contracts/baseline, local Laya shadow - trial, and measured keep-off or reviewed adoption. It prioritizes routing + amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped + Jev pilot, and local Laya fallback plus measured adoption. It prioritizes routing hints, skill/tool shortlists and context ranking, followed by failure triage, review attention and outcome labels. No weights/inference/API spending or - runtime routing changes occurred. Jev comparison needs separate API/data - authorization; classifier suggestions never become permission/acceptance - authority. M5 remains the immediate implementation priority. + runtime routing changes occurred. The user has authorized one Jev request + using only the synthetic fixture, no retries and at most $0.01. The tracked + fixture/probe and five focused offline tests are ready; the complete offline + gate is 231 core tests discovered (suite OK, 2 optional SDK skips) and 220 + Bash assertions. The live call is + blocked because the TypeSafe console is at login and no `TYPESAFE_API_KEY` + exists. Classifier suggestions never become permission/acceptance authority. + After the bounded probe, M5 remains the implementation priority. ## Completed and preserved @@ -129,7 +134,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 226 discovered at `30b98df`; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | +| Python core discovery | 231 discovered through the prepared Jev pilot; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -178,6 +183,9 @@ advertised as verified. 4. When M5's evidence shape is stable, execute the small M6 decision-helper work packages alongside other independently ready M6 work. Keep experiments off by default and preserve all existing M4/M5/live acceptance gates. + Exception already authorized: once TypeSafe login/API-key setup is complete, + run the prepared one-request synthetic Jev pilot immediately, save the + private receipt outside Git, and update only redacted aggregate evidence. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. diff --git a/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md index f8db0ff..c3edf66 100644 --- a/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md +++ b/docs/plans/engineering-team/SELECTION-AND-COUNCIL.md @@ -65,7 +65,9 @@ Reports explain selections, excluded alternatives, explicit overrides, escalatio The [model lifecycle amendment](MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md) adds stable aliases, automatic discovery and bounded qualification. After calibration, an enabled `guarded_auto` policy can promote tested bindings without per-release manual edits. The reviewed policy remains the authority; discovery or council votes alone cannot promote a candidate. The September 26 [Jev/Laya decision-helper amendment](DECISION-CLASSIFIERS.md) -adds a default-off, local-first evaluation within M6. Frozen suggestions may +adds a default-off evaluation within M6: one tightly capped Jev synthetic +smoke first, with Laya as the local fallback if access, cost or measured quality +justifies its heavier setup. Frozen suggestions may inform only a reviewed policy after use-case-specific held-out validation. The router remains deterministic: validate requirements, pins and eligible profiles before considering suggestions; invalid/uncertain/missing output uses diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index 92b66c1..265445f 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -37,7 +37,9 @@ Also read the [native adapters and model lifecycle amendment](MODEL-LIFECYCLE-AN The September 26 [Jev/Laya evaluation and decision-helper amendment](DECISION-CLASSIFIERS.md) adds optional M6 experiments for routing hints, skill selection and context ranking, with further bounded uses prioritized. It changes no runtime defaults -and does not delay M5 or authorize paid API usage. +and does not delay M5. The only current hosted authorization is the explicitly +capped one-request, $0.01 synthetic Jev pilot; no purchase, retry or private +task upload is authorized. ## Copyable execution brief diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 7555680..48633e2 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -263,26 +263,28 @@ { "id": "M6-D1", "title": "Optional typed decision contract and frozen evaluation baseline", - "status": "pending", + "status": "in_progress", "specification": "DECISION-CLASSIFIERS.md", - "evidence": [] + "evidence": [], + "checkpoint": "One-request synthetic Jev fixture, strict response validator and five offline probe tests are ready; core run/cache integration remains pending" }, { "id": "M6-D2", - "title": "Pinned local Laya shadow trial for task/profile and skill hints", + "title": "Capped Jev synthetic pilot for task/profile and skill hints", "depends_on": ["M6-D1"], - "status": "pending", + "status": "blocked", "specification": "DECISION-CLASSIFIERS.md", - "evidence": [] + "evidence": [], + "blocker": "The TypeSafe console is at login and no TYPESAFE_API_KEY is available; one billable request, no retries, synthetic input and a $0.01 ceiling are prepared" }, { "id": "M6-D3", - "title": "Measured keep-off or reviewed advisory adoption decision", + "title": "Triggered local Laya fallback and measured adoption decision", "depends_on": ["M6-D2"], "status": "pending", "specification": "DECISION-CLASSIFIERS.md", "evidence": [], - "hosted_comparison": "Optional Jev evaluation requires separate explicit API spending and data-sharing authorization; local evaluation does not depend on it" + "fallback_rule": "Use pinned Laya locally if Jev access requires a larger purchase, price/usage exceeds the frozen cap, or measured quality/latency fails the predeclared gate" } ], "evidence": [], diff --git a/docs/plans/engineering-team/experiments/jev-pilot-v1.json b/docs/plans/engineering-team/experiments/jev-pilot-v1.json new file mode 100644 index 0000000..2159858 --- /dev/null +++ b/docs/plans/engineering-team/experiments/jev-pilot-v1.json @@ -0,0 +1,130 @@ +{ + "schema_version": 1, + "experiment_id": "jev-devsquad-routing-pilot-v1", + "status": "ready_for_live_run", + "model": "jev-1.13.0", + "endpoint": "https://api.typesafe.ai/v1/systemone", + "pricing": { + "checked_at": "2026-09-26", + "usd_per_million_input_tokens": 0.042, + "max_cost_usd": 0.01 + }, + "budget": { + "max_billable_requests": 1, + "retries": 0, + "timeout_seconds": 30, + "data_class": "synthetic_public_fixture" + }, + "purposes": [ + { + "id": "task_family", + "instructions": "Classify the specified engineering task by its primary purpose. Use uncertain when the task lacks enough information.", + "criteria": { + "bug_fix": "Correct a known faulty behavior or regression.", + "implementation": "Build or materially change a feature.", + "investigation": "Diagnose a cause or analyze evidence without an already-defined fix.", + "review_validation": "Review, test, audit, or verify existing work.", + "documentation": "Primarily change explanatory documentation.", + "operations": "Operate or recover development tooling, source control, packaging, or deployment state.", + "uncertain": "The request is too vague or conflicting to classify safely." + } + }, + { + "id": "execution_tier", + "instructions": "Choose the least costly sufficient execution tier for the specified task. This is advisory only; permissions and verified capabilities are filtered separately.", + "criteria": { + "economy_read": "Low-risk read-only or documentation reasoning with clear scope.", + "standard_write": "Bounded, well-specified code or documentation change with ordinary tests.", + "frontier_analysis": "Difficult diagnosis, architecture, concurrency, or security reasoning without an immediately safe write.", + "frontier_write": "High-risk or complex implementation requiring strong reasoning and mandatory independent verification.", + "lead_clarification": "The task should not be delegated until scope, requirements, or authority are clarified." + } + }, + { + "id": "specialist_skill", + "instructions": "Choose at most one specialist skill that is important for the specified task. Choose none when ordinary engineering instructions are sufficient.", + "criteria": { + "git_safety": "Recovery, branch, worktree, history, or uncommitted-work safety is central.", + "security_review": "Security boundaries, authentication, authorization, secrets, or vulnerability analysis is central.", + "browser_qa": "A browser UI must be operated or visually validated.", + "data_analysis": "Structured data quality, metrics, statistical analysis, or reporting is central.", + "none": "No listed specialist skill is central.", + "uncertain": "The request lacks enough information to select safely." + } + } + ], + "cases": [ + { + "id": "T01", + "task": "Correct two spelling mistakes in a committed README section. No commands beyond the documentation link checker are required.", + "expected": { + "task_family": "documentation", + "execution_tier": "standard_write", + "specialist_skill": "none" + } + }, + { + "id": "T02", + "task": "Fix a deterministic parser off-by-one error in src/parser.py. The failing unit test, allowed file, and acceptance result are supplied.", + "expected": { + "task_family": "bug_fix", + "execution_tier": "standard_write", + "specialist_skill": "none" + } + }, + { + "id": "T03", + "task": "Diagnose an intermittent SQLite lease race observed only across two processes. Preserve all evidence and do not change code until the cause is established.", + "expected": { + "task_family": "investigation", + "execution_tier": "frontier_analysis", + "specialist_skill": "none" + } + }, + { + "id": "T04", + "task": "Repair an authorization bypass in the worker handoff endpoint. The change touches the access-control boundary and must receive security review and mandatory tests.", + "expected": { + "task_family": "bug_fix", + "execution_tier": "frontier_write", + "specialist_skill": "security_review" + } + }, + { + "id": "T05", + "task": "Verify a browser settings workflow at desktop and mobile breakpoints, including the visible success state. Do not modify the application.", + "expected": { + "task_family": "review_validation", + "execution_tier": "economy_read", + "specialist_skill": "browser_qa" + } + }, + { + "id": "T06", + "task": "Make the project better using whichever models and tools seem best.", + "expected": { + "task_family": "uncertain", + "execution_tier": "lead_clarification", + "specialist_skill": "uncertain" + } + }, + { + "id": "T07", + "task": "Analyze a CSV of task outcomes for missing values, routing accuracy, rework rate, and confidence intervals. Produce a source-backed report only.", + "expected": { + "task_family": "investigation", + "execution_tier": "frontier_analysis", + "specialist_skill": "data_analysis" + } + }, + { + "id": "T08", + "task": "Recover valuable uncommitted changes after work continued on the wrong Git branch. Preserve every edit and do not rewrite shared history.", + "expected": { + "task_family": "operations", + "execution_tier": "frontier_write", + "specialist_skill": "git_safety" + } + } + ] +} diff --git a/test/core/probes/jev_decision_eval.py b/test/core/probes/jev_decision_eval.py new file mode 100644 index 0000000..3a6c1dd --- /dev/null +++ b/test/core/probes/jev_decision_eval.py @@ -0,0 +1,364 @@ +#!/usr/bin/env python3 +"""Run one bounded, synthetic Jev decision-classifier pilot. + +The probe is deliberately separate from the DevSquad runtime. It makes exactly +one billable request, performs no retries, accepts the API key only through the +environment, and writes a redacted result containing no task text. +""" + +from __future__ import annotations + +import argparse +from datetime import datetime, timezone +import hashlib +import json +import math +import os +from pathlib import Path +import ssl +import sys +import tempfile +import time +from typing import Any +from urllib.error import HTTPError, URLError +from urllib.request import Request, urlopen + + +ROOT = Path(__file__).resolve().parents[3] +DEFAULT_SPEC = ( + ROOT + / "docs" + / "plans" + / "engineering-team" + / "experiments" + / "jev-pilot-v1.json" +) +CONTEXT_LIMIT_TOKENS = 64_000 +OFFICIAL_ENDPOINT = "https://api.typesafe.ai/v1/systemone" +PINNED_MODEL = "jev-1.13.0" +MAX_RESPONSE_BYTES = 1_048_576 + + +class ProbeError(RuntimeError): + """A safe, user-facing probe failure.""" + + +def load_spec(path: Path) -> dict[str, Any]: + with path.open(encoding="utf-8") as handle: + spec = json.load(handle) + required = { + "schema_version", + "experiment_id", + "status", + "model", + "endpoint", + "pricing", + "budget", + "purposes", + "cases", + } + if set(spec) != required: + raise ProbeError("pilot spec fields do not match the version-1 contract") + if spec["schema_version"] != 1: + raise ProbeError("unsupported pilot spec version") + if spec["status"] != "ready_for_live_run": + raise ProbeError("pilot spec is not approved for a live run") + if spec["endpoint"] != OFFICIAL_ENDPOINT or spec["model"] != PINNED_MODEL: + raise ProbeError("pilot endpoint and model must remain pinned") + pricing = spec["pricing"] + if not isinstance(pricing, dict) or set(pricing) != { + "checked_at", + "usd_per_million_input_tokens", + "max_cost_usd", + }: + raise ProbeError("pilot pricing fields are invalid") + if pricing["checked_at"] != "2026-09-26": + raise ProbeError("pilot pricing must be rechecked before changing its date") + if not all( + type(pricing[name]) in {int, float} + and math.isfinite(pricing[name]) + and pricing[name] > 0 + for name in ("usd_per_million_input_tokens", "max_cost_usd") + ): + raise ProbeError("pilot pricing values must be finite and positive") + if pricing["max_cost_usd"] != 0.01: + raise ProbeError("pilot cost ceiling must remain exactly $0.01") + if spec["budget"] != { + "max_billable_requests": 1, + "retries": 0, + "timeout_seconds": 30, + "data_class": "synthetic_public_fixture", + }: + raise ProbeError("pilot budget must remain one request with no retries") + if not spec["cases"] or not spec["purposes"]: + raise ProbeError("pilot requires cases and purposes") + case_ids = [case.get("id") for case in spec["cases"]] + purpose_ids = [purpose.get("id") for purpose in spec["purposes"]] + if len(case_ids) != len(set(case_ids)) or not all( + isinstance(value, str) and value for value in case_ids + ): + raise ProbeError("case IDs must be unique non-empty strings") + if len(purpose_ids) != len(set(purpose_ids)) or not all( + isinstance(value, str) and value for value in purpose_ids + ): + raise ProbeError("purpose IDs must be unique non-empty strings") + for case in spec["cases"]: + if set(case) != {"id", "task", "expected"}: + raise ProbeError(f"case {case.get('id')} has invalid fields") + if set(case["expected"]) != set(purpose_ids): + raise ProbeError(f"case {case['id']} lacks an expected purpose label") + purpose_map = {item["id"]: item for item in spec["purposes"]} + for purpose_id, purpose in purpose_map.items(): + if set(purpose) != {"id", "instructions", "criteria"}: + raise ProbeError(f"purpose {purpose_id} has invalid fields") + if not isinstance(purpose["instructions"], str) or not purpose["instructions"]: + raise ProbeError(f"purpose {purpose_id} lacks instructions") + if ( + not isinstance(purpose["criteria"], dict) + or len(purpose["criteria"]) < 2 + or not all( + isinstance(key, str) + and key + and isinstance(value, str) + and value + for key, value in purpose["criteria"].items() + ) + ): + raise ProbeError(f"purpose {purpose_id} criteria are invalid") + for case in spec["cases"]: + if not isinstance(case["task"], str) or not case["task"]: + raise ProbeError(f"case {case['id']} lacks task text") + for purpose_id, expected in case["expected"].items(): + if expected not in purpose_map[purpose_id]["criteria"]: + raise ProbeError(f"case {case['id']} has an unknown expected label") + return spec + + +def build_request(spec: dict[str, Any]) -> dict[str, Any]: + state = [ + {"task_id": case["id"], "task": case["task"]} + for case in spec["cases"] + ] + questions: dict[str, Any] = {} + for case in spec["cases"]: + for purpose in spec["purposes"]: + key = f"{case['id']}__{purpose['id']}" + questions[key] = { + "type": "choice", + "instructions": ( + f"For task_id {case['id']} only: {purpose['instructions']}" + ), + "criteria": purpose["criteria"], + } + return {"state": state, "model": spec["model"], "questions": questions} + + +def _finite_probability(value: Any) -> bool: + return ( + type(value) in {int, float} + and math.isfinite(value) + and 0.0 <= float(value) <= 1.0 + ) + + +def validate_response( + spec: dict[str, Any], request_body: dict[str, Any], response: Any +) -> dict[str, Any]: + if not isinstance(response, dict) or set(response) != {"model", "answers", "usage"}: + raise ProbeError("Jev response envelope is invalid") + if response["model"] != spec["model"]: + raise ProbeError("Jev response model does not match the concrete pin") + answers = response["answers"] + if not isinstance(answers, dict) or set(answers) != set(request_body["questions"]): + raise ProbeError("Jev answer IDs do not match the frozen questions") + usage = response["usage"] + if not isinstance(usage, dict) or set(usage) != {"input_tokens", "output_tokens"}: + raise ProbeError("Jev usage is invalid") + if not all(type(usage[name]) is int and usage[name] >= 0 for name in usage): + raise ProbeError("Jev usage counts must be non-negative integers") + + results = [] + purpose_map = {item["id"]: item for item in spec["purposes"]} + for case in spec["cases"]: + predictions: dict[str, Any] = {} + for purpose_id, purpose in purpose_map.items(): + key = f"{case['id']}__{purpose_id}" + answer = answers[key] + criteria = set(purpose["criteria"]) + if not isinstance(answer, dict) or set(answer) != { + "type", + "choice", + "confidence", + "probabilities", + }: + raise ProbeError(f"answer {key} has invalid fields") + probabilities = answer["probabilities"] + if answer["type"] != "choice" or answer["choice"] not in criteria: + raise ProbeError(f"answer {key} has an unknown choice") + if not _finite_probability(answer["confidence"]): + raise ProbeError(f"answer {key} has invalid confidence") + if not isinstance(probabilities, dict) or set(probabilities) != criteria: + raise ProbeError(f"answer {key} probability labels are invalid") + if not all(_finite_probability(value) for value in probabilities.values()): + raise ProbeError(f"answer {key} probabilities are invalid") + if not math.isclose(sum(probabilities.values()), 1.0, abs_tol=0.02): + raise ProbeError(f"answer {key} probabilities do not sum to one") + predictions[purpose_id] = { + "expected": case["expected"][purpose_id], + "choice": answer["choice"], + "correct": answer["choice"] == case["expected"][purpose_id], + "confidence": answer["confidence"], + "probabilities": probabilities, + } + results.append({"case_id": case["id"], "predictions": predictions}) + return {"model": response["model"], "usage": usage, "cases": results} + + +def summarize( + spec: dict[str, Any], + validated: dict[str, Any], + elapsed_ms: int, + request_sha256: str, +) -> dict[str, Any]: + per_purpose: dict[str, dict[str, int]] = { + purpose["id"]: {"correct": 0, "total": 0} + for purpose in spec["purposes"] + } + for case in validated["cases"]: + for purpose_id, prediction in case["predictions"].items(): + per_purpose[purpose_id]["total"] += 1 + per_purpose[purpose_id]["correct"] += int(prediction["correct"]) + for counts in per_purpose.values(): + counts["accuracy_percent"] = round( + 100.0 * counts["correct"] / counts["total"], 1 + ) + input_tokens = validated["usage"]["input_tokens"] + rate = spec["pricing"]["usd_per_million_input_tokens"] + cost = input_tokens * rate / 1_000_000 + return { + "schema_version": 1, + "experiment_id": spec["experiment_id"], + "executed_at": datetime.now(timezone.utc).isoformat(), + "data_class": spec["budget"]["data_class"], + "request_sha256": request_sha256, + "billable_requests": 1, + "requested_model": spec["model"], + "observed_model": validated["model"], + "elapsed_ms": elapsed_ms, + "usage": validated["usage"], + "pricing": { + "usd_per_million_input_tokens": rate, + "estimated_cost_usd": round(cost, 8), + "max_cost_usd": spec["pricing"]["max_cost_usd"], + }, + "per_purpose": per_purpose, + "cases": validated["cases"], + "interpretation": ( + "Synthetic smoke result only; it does not establish a production " + "routing improvement or authorize advisory mode." + ), + } + + +def write_json(path: Path, value: dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with tempfile.NamedTemporaryFile( + "w", encoding="utf-8", dir=path.parent, delete=False + ) as handle: + json.dump(value, handle, indent=2, sort_keys=True) + handle.write("\n") + temporary = Path(handle.name) + temporary.replace(path) + + +def execute(spec: dict[str, Any], request_body: dict[str, Any]) -> dict[str, Any]: + key = os.environ.get("TYPESAFE_API_KEY") + if not key: + raise ProbeError("TYPESAFE_API_KEY is not set") + encoded = json.dumps(request_body, separators=(",", ":")).encode("utf-8") + request = Request( + spec["endpoint"], + data=encoded, + headers={ + "Authorization": f"Bearer {key}", + "Content-Type": "application/json", + "User-Agent": "devsquad-jev-pilot/1", + }, + method="POST", + ) + start = time.monotonic() + try: + with urlopen( + request, + timeout=spec["budget"]["timeout_seconds"], + context=ssl.create_default_context(), + ) as response: + raw = response.read(MAX_RESPONSE_BYTES + 1) + if len(raw) > MAX_RESPONSE_BYTES: + raise ProbeError("TypeSafe API response exceeded the size limit") + payload = json.loads(raw.decode("utf-8")) + except HTTPError as exc: + raise ProbeError(f"TypeSafe API returned HTTP {exc.code}; no retry attempted") from None + except URLError as exc: + raise ProbeError(f"TypeSafe API was unreachable ({exc.reason}); no retry attempted") from None + except (TimeoutError, json.JSONDecodeError): + raise ProbeError("TypeSafe API timed out or returned invalid JSON; no retry attempted") from None + elapsed_ms = round((time.monotonic() - start) * 1000) + validated = validate_response(spec, request_body, payload) + request_sha256 = hashlib.sha256(encoded).hexdigest() + return summarize(spec, validated, elapsed_ms, request_sha256) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--spec", type=Path, default=DEFAULT_SPEC) + parser.add_argument("--execute", action="store_true") + parser.add_argument("--output", type=Path) + args = parser.parse_args(argv) + try: + spec = load_spec(args.spec) + request_body = build_request(spec) + price = spec["pricing"]["usd_per_million_input_tokens"] + max_cost = spec["pricing"]["max_cost_usd"] + documented_max_cost = CONTEXT_LIMIT_TOKENS * price / 1_000_000 + if documented_max_cost > max_cost: + raise ProbeError( + "one maximum-context request exceeds the frozen cost ceiling; " + "stop and evaluate Laya" + ) + dry_run = { + "experiment_id": spec["experiment_id"], + "cases": len(spec["cases"]), + "questions": len(request_body["questions"]), + "billable_requests": 1, + "retries": 0, + "request_bytes": len( + json.dumps(request_body, separators=(",", ":")).encode("utf-8") + ), + "request_sha256": hashlib.sha256( + json.dumps(request_body, separators=(",", ":")).encode("utf-8") + ).hexdigest(), + "documented_max_request_cost_usd": round(documented_max_cost, 8), + "cost_ceiling_usd": max_cost, + "data_class": spec["budget"]["data_class"], + } + if not args.execute: + print(json.dumps(dry_run, indent=2, sort_keys=True)) + return 0 + if args.output is None: + raise ProbeError("--output is required for a live run") + if args.output.resolve().is_relative_to(ROOT): + raise ProbeError("live output must remain outside the Git repository") + result = execute(spec, request_body) + write_json(args.output, result) + print(json.dumps({**dry_run, "result": str(args.output)}, indent=2, sort_keys=True)) + if result["pricing"]["estimated_cost_usd"] > max_cost: + raise ProbeError("actual reported usage exceeded the cost ceiling; switch to Laya") + return 0 + except ProbeError as exc: + print(f"JEV_PROBE_ERROR: {exc}", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/test/core/test_jev_decision_probe.py b/test/core/test_jev_decision_probe.py new file mode 100644 index 0000000..fbd1731 --- /dev/null +++ b/test/core/test_jev_decision_probe.py @@ -0,0 +1,128 @@ +from __future__ import annotations + +import importlib.util +import json +import math +from pathlib import Path +import subprocess +import sys +import unittest + + +ROOT = Path(__file__).resolve().parents[2] +PROBE = ROOT / "test" / "core" / "probes" / "jev_decision_eval.py" +SPEC = ( + ROOT + / "docs" + / "plans" + / "engineering-team" + / "experiments" + / "jev-pilot-v1.json" +) +MODULE_SPEC = importlib.util.spec_from_file_location("jev_decision_eval", PROBE) +assert MODULE_SPEC and MODULE_SPEC.loader +jev = importlib.util.module_from_spec(MODULE_SPEC) +MODULE_SPEC.loader.exec_module(jev) + + +class JevDecisionProbeTest(unittest.TestCase): + def setUp(self): + self.spec = jev.load_spec(SPEC) + self.request = jev.build_request(self.spec) + + def valid_response(self): + answers = {} + purpose_map = {item["id"]: item for item in self.spec["purposes"]} + for case in self.spec["cases"]: + for purpose_id, purpose in purpose_map.items(): + expected = case["expected"][purpose_id] + labels = list(purpose["criteria"]) + remaining = (1.0 - 0.85) / (len(labels) - 1) + probabilities = { + label: 0.85 if label == expected else remaining + for label in labels + } + answers[f"{case['id']}__{purpose_id}"] = { + "type": "choice", + "choice": expected, + "confidence": 0.8, + "probabilities": probabilities, + } + return { + "model": "jev-1.13.0", + "answers": answers, + "usage": {"input_tokens": 2100, "output_tokens": 420}, + } + + def test_frozen_probe_is_one_request_and_dry_run_needs_no_key(self): + expected_questions = len(self.spec["cases"]) * len(self.spec["purposes"]) + self.assertEqual(len(self.request["questions"]), expected_questions) + completed = subprocess.run( + [sys.executable, str(PROBE), "--spec", str(SPEC)], + check=True, + capture_output=True, + text=True, + ) + dry_run = json.loads(completed.stdout) + self.assertEqual(dry_run["billable_requests"], 1) + self.assertEqual(dry_run["retries"], 0) + self.assertLessEqual( + dry_run["documented_max_request_cost_usd"], + dry_run["cost_ceiling_usd"], + ) + + def test_valid_response_is_redacted_and_scored(self): + validated = jev.validate_response( + self.spec, self.request, self.valid_response() + ) + summary = jev.summarize( + self.spec, validated, elapsed_ms=123, request_sha256="a" * 64 + ) + self.assertEqual(summary["billable_requests"], 1) + self.assertEqual(summary["per_purpose"]["task_family"]["accuracy_percent"], 100.0) + self.assertNotIn("Correct two spelling mistakes", json.dumps(summary)) + self.assertTrue(math.isclose(summary["pricing"]["estimated_cost_usd"], 0.0000882)) + + def test_unknown_choice_and_non_finite_probability_are_rejected(self): + response = self.valid_response() + key = next(iter(response["answers"])) + response["answers"][key]["choice"] = "not-a-label" + with self.assertRaisesRegex(jev.ProbeError, "unknown choice"): + jev.validate_response(self.spec, self.request, response) + + response = self.valid_response() + key = next(iter(response["answers"])) + label = next(iter(response["answers"][key]["probabilities"])) + response["answers"][key]["probabilities"][label] = math.nan + with self.assertRaisesRegex(jev.ProbeError, "probabilities are invalid"): + jev.validate_response(self.spec, self.request, response) + + def test_live_mode_requires_output_before_network(self): + completed = subprocess.run( + [sys.executable, str(PROBE), "--spec", str(SPEC), "--execute"], + capture_output=True, + text=True, + ) + self.assertEqual(completed.returncode, 2) + self.assertIn("--output is required", completed.stderr) + + def test_live_result_cannot_be_written_into_the_repository(self): + completed = subprocess.run( + [ + sys.executable, + str(PROBE), + "--spec", + str(SPEC), + "--execute", + "--output", + str(ROOT / "jev-result.json"), + ], + capture_output=True, + text=True, + ) + self.assertEqual(completed.returncode, 2) + self.assertIn("outside the Git repository", completed.stderr) + + +if __name__ == "__main__": + unittest.main() From df12b7384552eedf799d5a1dd1a4635e0c8dda0d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sat, 26 Sep 2026 10:27:25 -0700 Subject: [PATCH 088/197] WIP checkpoint: record Jev pilot readiness and blocker (2026-09-26 10:27) --- docs/plans/engineering-team/RESUME.md | 3 ++- docs/plans/engineering-team/backlog.json | 24 ++++++++++++++++++++++-- 2 files changed, 24 insertions(+), 3 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 4acd8aa..bce246e 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -117,7 +117,8 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne review attention and outcome labels. No weights/inference/API spending or runtime routing changes occurred. The user has authorized one Jev request using only the synthetic fixture, no retries and at most $0.01. The tracked - fixture/probe and five focused offline tests are ready; the complete offline + fixture/probe checkpoint is committed at `70e59cb` and five focused offline + tests are ready; the complete offline gate is 231 core tests discovered (suite OK, 2 optional SDK skips) and 220 Bash assertions. The live call is blocked because the TypeSafe console is at login and no `TYPESAFE_API_KEY` diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 48633e2..cc1f3c4 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -265,7 +265,17 @@ "title": "Optional typed decision contract and frozen evaluation baseline", "status": "in_progress", "specification": "DECISION-CLASSIFIERS.md", - "evidence": [], + "evidence": [ + { + "kind": "experiment_preparation", + "revision": "70e59cb", + "command_or_action": "One-request dry run plus 5 focused probe tests, 231-test complete core discovery with ResourceWarning promoted to error, and 220 Bash assertions", + "outcome": "The pinned Jev 1.13 synthetic pilot has 8 cases and 24 typed choices, a $0.01 ceiling, no retry, strict model/probability/usage validation and private-output enforcement; no provider request ran because TypeSafe login/API key is unavailable", + "artifact": "experiments/jev-pilot-v1.json", + "recorded_at": "2026-09-26T17:26:40Z", + "availability": "tracked_fixture_and_tests" + } + ], "checkpoint": "One-request synthetic Jev fixture, strict response validator and five offline probe tests are ready; core run/cache integration remains pending" }, { @@ -287,7 +297,17 @@ "fallback_rule": "Use pinned Laya locally if Jev access requires a larger purchase, price/usage exceeds the frozen cap, or measured quality/latency fails the predeclared gate" } ], - "evidence": [], + "evidence": [ + { + "kind": "early_experiment_checkpoint", + "revision": "70e59cb", + "command_or_action": "Prepared and offline-validated the explicitly authorized capped Jev pilot before M6 runtime integration", + "outcome": "Synthetic probe mechanics are ready and ordinary runtime behavior is unchanged; the billable live smoke is blocked on TypeSafe login/API-key setup and cannot count as classifier quality evidence", + "artifact": "experiments/jev-pilot-v1.json", + "recorded_at": "2026-09-26T17:26:40Z", + "availability": "tracked_fixture_and_tests" + } + ], "blocker": null }, { From a1199c6ca640b18c9ee0a487be6ea21e9cd58ab1 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sat, 26 Sep 2026 23:44:41 -0700 Subject: [PATCH 089/197] feat: complete delivery lead dispositions --- plugin/core/src/devsquad/reports.py | 58 +++++++++++++--- plugin/core/src/devsquad/service.py | 18 ++--- test/core/test_delivery_workflow.py | 103 +++++++++++++++++++++++++++- 3 files changed, 162 insertions(+), 17 deletions(-) diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index df465ba..5ab21de 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -138,10 +138,13 @@ def build_handoff_reports( if hashlib.sha256(packet_json.encode()).hexdigest() != packet_sha256: raise ContractError("handoff report packet hash is invalid") json_name, markdown_name = handoff_report_names(sequence) + workflow = packet.get("workflow") + if workflow not in {"branch-review", "issue-delivery"}: + raise ContractError("handoff report workflow is invalid") report = { "schema_version": 1, "run_id": run_id, - "workflow": "branch-review", + "workflow": workflow, "state": "awaiting_host", "created_at": created_at, "handoff_id": handoff_id, @@ -157,7 +160,7 @@ def build_handoff_reports( } review = packet.get("review") if isinstance(packet.get("review"), dict) else {} lines = [ - "# DevSquad branch review handoff", + f"# DevSquad {workflow} handoff", "", f"- Run: `{run_id}`", f"- Handoff: `{handoff_id}`", @@ -386,6 +389,9 @@ def _history( dispositions = [] prior_sequence = 0 candidate = None + workflow = snapshot.get("task", {}).get("workflow") + if workflow not in {"branch-review", "issue-delivery"}: + raise ContractError("review report workflow is invalid") for index, entry in enumerate(entries): required = { "handoff_id", "sequence", "packet", "packet_sha256", "decision", @@ -410,7 +416,7 @@ def _history( ) if candidate is None: candidate = identity - elif identity != candidate: + elif identity != candidate and workflow == "branch-review": raise ContractError("branch review report history changes the candidate") if index < len(entries) - 1 and decision["disposition"] != "revise": raise ContractError("only a revision may precede another review attempt") @@ -450,7 +456,7 @@ def project_branch_review_history( def _markdown(receipt: dict[str, Any]) -> str: review = receipt["review"] lines = [ - "# DevSquad branch review", + f"# DevSquad {receipt['workflow']}", "", f"- Run: `{receipt['run_id']}`", f"- State: `{receipt['state']}`", @@ -515,9 +521,12 @@ def build_terminal_reports( if not isinstance(run_id, str) or not run_id: raise ContractError("report run id is invalid") if state not in {"succeeded", "failed"}: - raise ContractError("branch review report state is invalid") + raise ContractError("review report state is invalid") if not isinstance(snapshot, dict) or not isinstance(snapshot.get("routing"), dict): raise ContractError("branch review report snapshot is invalid") + workflow = snapshot.get("task", {}).get("workflow") + if workflow not in {"branch-review", "issue-delivery"}: + raise ContractError("review report workflow is invalid") attempts, dispositions = _history(history, snapshot) final_packet = validate_branch_review_handoff(history[-1]["packet"], snapshot) final_decision = _decision(history[-1]["decision"]) @@ -581,14 +590,42 @@ def build_terminal_reports( failed_leads = [attempt for attempt in failed if attempt["role"] == "lead"] reviewer_attempts = failed_reviewers + attempts lead_attempts = failed_leads + lead_attempts - all_attempts = reviewer_attempts + lead_attempts + implementation_attempts = [] + if workflow == "issue-delivery": + iterations = snapshot.get("delivery_iterations") + if not isinstance(iterations, list) or not iterations: + raise ContractError("delivery report has no candidate iterations") + for expected_iteration, iteration in enumerate(iterations, 1): + if (not isinstance(iteration, dict) + or iteration.get("iteration") != expected_iteration + or not isinstance(iteration.get("candidate"), dict) + or not isinstance(iteration.get("implementation"), dict) + or not isinstance(iteration["implementation"].get("attempt"), dict)): + raise ContractError("delivery report candidate iteration is invalid") + artifact_name = iteration["candidate"].get("implementation_artifact") + if (not isinstance(artifact_name, str) + or not artifact_name.startswith("implementation-attempt-") + or not artifact_name.endswith(".json")): + raise ContractError("delivery report implementation artifact is invalid") + implementation_attempts.append({ + "id": artifact_name[ + len("implementation-attempt-"):-len(".json") + ], + "status": "succeeded", + "sequence": expected_iteration, + **iteration["implementation"]["attempt"], + "summary": iteration["implementation"].get("summary"), + "candidate": iteration["candidate"], + "evidence_refs": [artifact_name], + }) + all_attempts = implementation_attempts + reviewer_attempts + lead_attempts all_native_counts = [ attempt["native_model_requests"] for attempt in all_attempts ] receipt = { "schema_version": 1, "run_id": run_id, - "workflow": "branch-review", + "workflow": workflow, "state": state, "completed_at": completed_at, "candidate": { @@ -601,7 +638,12 @@ def build_terminal_reports( "checks": final_packet["checks"], "evaluation": final_packet["evaluation"], "criteria": final_packet["evaluation"]["criteria"], - "attempts": reviewer_attempts, + "attempts": ( + implementation_attempts + reviewer_attempts + if workflow == "issue-delivery" else reviewer_attempts + ), + **({"delivery_iterations": snapshot["delivery_iterations"]} + if workflow == "issue-delivery" else {}), "dispositions": dispositions, "revisions": { "requested": sum( diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 1b88202..905625b 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -1256,21 +1256,23 @@ def handoff_complete( try: run = store.run(run_id) handoff = store.handoff_snapshot_by_id(run_id, decoded.handoff_id) - branch_review = handoff.packet.get("workflow") == "branch-review" - snapshot = self._review_snapshot(run) if branch_review else None - if (branch_review and snapshot["task"]["lead"]["mode"] == "headless" + managed_review = handoff.packet.get("workflow") in { + "branch-review", "issue-delivery", + } + snapshot = self._review_snapshot(run) if managed_review else None + if (managed_review and snapshot["task"]["lead"]["mode"] == "headless" and not decoded.owner_id.startswith("headless-lead:")): raise ConflictError( - "headless branch review does not accept a host completion" + "headless review does not accept a host completion" ) - if branch_review: + if managed_review: self._review_gate(store, run_id, handoff, snapshot, decision) submission = store.record_handoff_submission(run_id, decoded, decision) continuation = None - if branch_review: + if managed_review: entry = store.recorded_handoff_submission(run_id, decoded.handoff_id) if entry is None: - raise ConflictError("recorded branch review submission is missing") + raise ConflictError("recorded review submission is missing") continuation = self._continue_branch_review_submission( store, run_id, handoff, snapshot, entry, ) @@ -1296,7 +1298,7 @@ def handoff_complete( version, package, digest = launch self._spawn_daemon(run_id, version, package, digest) response["launched"] = True - elif branch_review: + elif managed_review: response["launched"] = False return response diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py index 0928207..652f590 100644 --- a/test/core/test_delivery_workflow.py +++ b/test/core/test_delivery_workflow.py @@ -14,7 +14,7 @@ from devsquad.contracts import ContractError from devsquad.service import Service -from devsquad.store import Store +from devsquad.store import Store, request_hash from devsquad.workspaces import ( freeze_delivery_candidate, prepare_delivery_workspace, @@ -223,6 +223,28 @@ def clean_review_fixture() -> dict[str, object]: "findings": [], } + @staticmethod + def decision( + packet: dict[str, object], + submission_id: str, + disposition: str, + reason: str, + ) -> dict[str, object]: + body = { + "schema_version": 1, + "submission_id": submission_id, + "disposition": disposition, + "reason": reason, + "evidence_refs": [ + { + "artifact_id": reference["artifact_id"], + "sha256": reference["sha256"], + } + for reference in packet["artifacts"] + ], + } + return {**body, "submission_hash": request_hash(body)} + def wait_for_candidate(self, service: Service, run_id: str) -> dict[str, object]: deadline = time.monotonic() + 15 while time.monotonic() < deadline: @@ -437,6 +459,85 @@ def test_exact_candidate_is_reviewed_checked_and_published_for_lead(self): ) self.assert_source_unchanged() + def test_host_accept_publishes_complete_delivery_receipt(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "accepted-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + status = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], status["version"], "fixture-host", + ) + packet = claimed["handoff"]["packet"] + completed = service.handoff_complete( + started["run_id"], + claimed["claim"], + self.decision(packet, "accept-delivery", "accept", "Candidate accepted."), + ) + self.assertEqual(completed["state"], "succeeded") + self.assertEqual(completed["continuation"]["action"], "terminal") + result = service.result(started["run_id"]) + receipt_artifact = next( + item for item in result["artifacts"] if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["workflow"], "issue-delivery") + self.assertEqual(receipt["candidate"]["sha256"], packet["candidate_sha256"]) + self.assertEqual( + [attempt["role"] for attempt in receipt["attempts"]], + ["implementer", "reviewer"], + ) + self.assertEqual(receipt["delivery_iterations"][0]["iteration"], 1) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(receipt["lead"]["disposition"], "accept") + self.assert_source_unchanged() + + def test_required_failure_blocks_delivery_accept_and_allows_reject(self): + task = self.delivery_task() + task["checks"][0] = { + **task["checks"][0], + "argv": ["python3", "-c", "raise SystemExit(1)"], + } + service = Service(self.runtime) + started = service.start( + task, + "rejected-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + status = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], status["version"], "fixture-host", + ) + packet = claimed["handoff"]["packet"] + with self.assertRaisesRegex(ContractError, "blocked by required evidence"): + service.handoff_complete( + started["run_id"], + claimed["claim"], + self.decision(packet, "blocked-accept", "accept", "Accept anyway."), + ) + completed = service.handoff_complete( + started["run_id"], + claimed["claim"], + self.decision(packet, "reject-delivery", "reject", "Required check failed."), + ) + self.assertEqual(completed["state"], "failed") + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "REVIEW_REJECTED") + self.assertEqual(receipt["evaluation"]["required_failures"], ["fixture-check"]) + self.assert_source_unchanged() + def test_live_implementer_cannot_be_resumed_into_a_second_writer(self): service = Service(self.runtime) started = service.start( From 4c768870199dbf516b9950ac1e311ccf6a20ea65 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sat, 26 Sep 2026 23:57:37 -0700 Subject: [PATCH 090/197] feat: execute bounded delivery revisions --- plugin/core/src/devsquad/delivery_worker.py | 7 + plugin/core/src/devsquad/detached.py | 10 + plugin/core/src/devsquad/reports.py | 70 ++++- plugin/core/src/devsquad/service.py | 95 +++++-- plugin/core/src/devsquad/store.py | 157 +++++++++++ plugin/core/src/devsquad/supervisor.py | 38 ++- plugin/core/src/devsquad/workflows.py | 23 +- plugin/core/src/devsquad/workspaces.py | 33 ++- test/core/test_delivery_workflow.py | 286 +++++++++++++++++++- 9 files changed, 659 insertions(+), 60 deletions(-) diff --git a/plugin/core/src/devsquad/delivery_worker.py b/plugin/core/src/devsquad/delivery_worker.py index ed7e2b6..2c87249 100644 --- a/plugin/core/src/devsquad/delivery_worker.py +++ b/plugin/core/src/devsquad/delivery_worker.py @@ -31,6 +31,13 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: if not isinstance(snapshot, dict): raise ContractError("delivery snapshot must be an object") fixture = snapshot.get("internal_implementation_fixture") + if isinstance(fixture, dict) and set(fixture) == {"iterations"}: + fixtures = fixture["iterations"] + index = len(snapshot.get("delivery_iterations", [])) + if (not isinstance(fixtures, list) or not fixtures + or index >= len(fixtures)): + raise ContractError("offline implementation fixture iteration is missing") + fixture = fixtures[index] if not isinstance(fixture, dict) or set(fixture) != {"writes", "delay_seconds"}: raise ContractError("offline implementation fixture is incomplete") writes, delay = fixture["writes"], fixture["delay_seconds"] diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index 71a09a2..b7f2413 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -47,6 +47,16 @@ def _profile_index( ) if reviewer is None: raise ConflictError("handoff reviewer attempt is missing") + if role == "implementer": + seen = False + used = 0 + for attempt in attempts: + if attempt["id"] == reviewer_id: + seen = True + elif (seen and attempt.get("role") == "implementer" + and _is_saved_fallback_failure(attempt)): + used += 1 + return used if role == "reviewer": seen = False used = 0 diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index 5ab21de..355f111 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -138,7 +138,7 @@ def build_handoff_reports( if hashlib.sha256(packet_json.encode()).hexdigest() != packet_sha256: raise ContractError("handoff report packet hash is invalid") json_name, markdown_name = handoff_report_names(sequence) - workflow = packet.get("workflow") + workflow = packet.get("workflow", "branch-review") if workflow not in {"branch-review", "issue-delivery"}: raise ContractError("handoff report workflow is invalid") report = { @@ -213,7 +213,8 @@ def build_early_terminal_reports( raise ContractError("report run id is invalid") if state not in {"failed", "cancelled"}: raise ContractError("early terminal report state is invalid") - if not isinstance(task, dict) or task.get("workflow") != "branch-review": + if (not isinstance(task, dict) + or task.get("workflow") not in {"branch-review", "issue-delivery"}): raise ContractError("early terminal report task is invalid") if snapshot is not None and not isinstance(snapshot, dict): raise ContractError("early terminal report snapshot is invalid") @@ -223,8 +224,15 @@ def build_early_terminal_reports( raise ContractError("early terminal report error is invalid") frozen = snapshot or {} + workflow = task["workflow"] workspace = frozen.get("workspace") + if (not isinstance(workspace, dict) and workflow == "issue-delivery" + and isinstance(frozen.get("delivery_iterations"), list) + and frozen["delivery_iterations"]): + workspace = frozen["delivery_iterations"][-1].get("workspace") workspace = workspace if isinstance(workspace, dict) else {} + delivery = frozen.get("delivery_workspace") + delivery = delivery if isinstance(delivery, dict) else {} projected = [ _unreferenced_artifact_projection(artifact) for artifact in run_artifacts ] @@ -277,14 +285,18 @@ def build_early_terminal_reports( receipt = { "schema_version": 1, "run_id": run_id, - "workflow": "branch-review", + "workflow": workflow, "state": state, "phase": phase, "completed_at": completed_at, "candidate": { "sha256": workspace.get("candidate_sha256"), - "base_oid": workspace.get("base_oid", frozen.get("base_oid")), - "target_oid": workspace.get("target_oid", frozen.get("target_oid")), + "base_oid": workspace.get( + "base_oid", delivery.get("baseline_oid", frozen.get("base_oid")), + ), + "target_oid": workspace.get( + "target_oid", delivery.get("baseline_oid", frozen.get("target_oid")), + ), }, "routing": frozen.get("routing"), "review": None, @@ -338,7 +350,7 @@ def build_early_terminal_reports( "error": error, } lines = [ - "# DevSquad branch review", + f"# DevSquad {workflow}", "", f"- Run: `{run_id}`", f"- State: `{state}`", @@ -379,6 +391,27 @@ def _decision(value: Any) -> dict[str, Any]: return value +def validate_saved_review_handoff( + packet: dict[str, Any], + snapshot: dict[str, Any], +) -> dict[str, Any]: + """Validate current or archived delivery evidence against its own workspace.""" + if snapshot.get("task", {}).get("workflow") != "issue-delivery": + return validate_branch_review_handoff(packet, snapshot) + candidate_sha256 = packet.get("candidate_sha256") + for iteration in snapshot.get("delivery_iterations", []): + if (isinstance(iteration, dict) + and iteration.get("candidate", {}).get("candidate_sha256") + == candidate_sha256 + and isinstance(iteration.get("workspace"), dict)): + historical = dict(snapshot) + historical["candidate"] = iteration["candidate"] + historical["workspace"] = iteration["workspace"] + historical["check_workspace"] = iteration.get("check_workspace") + return validate_branch_review_handoff(packet, historical) + raise ContractError("delivery handoff does not match a saved candidate iteration") + + def _history( entries: list[dict[str, Any]], snapshot: dict[str, Any], @@ -408,7 +441,7 @@ def _history( packet_json = canonical_json(entry["packet"]) if hashlib.sha256(packet_json.encode()).hexdigest() != entry["packet_sha256"]: raise ContractError("branch review report handoff hash is invalid") - packet = validate_branch_review_handoff(entry["packet"], snapshot) + packet = validate_saved_review_handoff(entry["packet"], snapshot) decision = _decision(entry["decision"]) validate_handoff_decision_evidence(decision, packet) identity = ( @@ -418,6 +451,16 @@ def _history( candidate = identity elif identity != candidate and workflow == "branch-review": raise ContractError("branch review report history changes the candidate") + elif identity == candidate and workflow == "issue-delivery": + raise ContractError("delivery report history repeats a candidate") + if workflow == "issue-delivery": + iterations = snapshot.get("delivery_iterations", []) + if (index >= len(iterations) + or packet["candidate_sha256"] + != iterations[index].get("candidate", {}).get( + "candidate_sha256" + )): + raise ContractError("delivery report candidate order is invalid") if index < len(entries) - 1 and decision["disposition"] != "revise": raise ContractError("only a revision may precede another review attempt") attempts.append({ @@ -528,7 +571,7 @@ def build_terminal_reports( if workflow not in {"branch-review", "issue-delivery"}: raise ContractError("review report workflow is invalid") attempts, dispositions = _history(history, snapshot) - final_packet = validate_branch_review_handoff(history[-1]["packet"], snapshot) + final_packet = validate_saved_review_handoff(history[-1]["packet"], snapshot) final_decision = _decision(history[-1]["decision"]) if (state == "succeeded") != (final_decision["disposition"] == "accept"): raise ContractError("terminal state and lead disposition disagree") @@ -577,16 +620,24 @@ def build_terminal_reports( "Reviewer output came from the explicit offline fixture; " "it is not live-provider evidence." ) + if "internal_implementation_fixture" in snapshot: + limitations.append( + "Implementation output came from the explicit offline fixture; " + "it is not live-provider evidence." + ) lead_attempts = [evidence["attempt"] for evidence in lead_evidence] failed = [] if failed_attempts is None else failed_attempts if not isinstance(failed, list) or not all( isinstance(attempt, dict) - and attempt.get("role") in {"reviewer", "lead"} + and attempt.get("role") in {"implementer", "reviewer", "lead"} for attempt in failed): raise ContractError("failed fallback attempts are invalid") failed_reviewers = [ attempt for attempt in failed if attempt["role"] == "reviewer" ] + failed_implementers = [ + attempt for attempt in failed if attempt["role"] == "implementer" + ] failed_leads = [attempt for attempt in failed if attempt["role"] == "lead"] reviewer_attempts = failed_reviewers + attempts lead_attempts = failed_leads + lead_attempts @@ -618,6 +669,7 @@ def build_terminal_reports( "candidate": iteration["candidate"], "evidence_refs": [artifact_name], }) + implementation_attempts = failed_implementers + implementation_attempts all_attempts = implementation_attempts + reviewer_attempts + lead_attempts all_native_counts = [ attempt["native_model_requests"] for attempt in all_attempts diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 905625b..8fccd60 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -25,6 +25,7 @@ build_early_terminal_reports, build_terminal_reports, project_branch_review_history, + validate_saved_review_handoff, ) from .router import capacity_with_live_reservations, load_routing from .store import ( @@ -74,7 +75,7 @@ def _preparation_failure_artifacts( snapshot: dict[str, Any] | None, error: dict[str, Any], ) -> list[dict[str, Any]] | None: - if task.get("workflow") != "branch-review": + if task.get("workflow") not in {"branch-review", "issue-delivery"}: return None reports = build_early_terminal_reports( run_id=run_id, @@ -128,7 +129,7 @@ def _saved_review_progress( for packet in packets)): packets.append(handoff.packet) packets_by_attempt = { - packet["attempt_id"]: validate_branch_review_handoff(packet, snapshot) + packet["attempt_id"]: validate_saved_review_handoff(packet, snapshot) for packet in packets } artifacts = store.artifacts_for_run(run_id) @@ -144,7 +145,21 @@ def _saved_review_progress( if attempt["id"] in failed_by_id: attempts.append(failed_by_id[attempt["id"]]) continue - if attempt.get("role") == "reviewer": + if attempt.get("role") == "implementer": + iteration = next(( + item for item in snapshot.get("delivery_iterations", []) + if item.get("candidate", {}).get("implementation_artifact") + == f"implementation-attempt-{attempt['id']}.json" + ), None) + if iteration is not None: + attempts.append({ + "id": attempt["id"], + "status": "succeeded", + **iteration["implementation"]["attempt"], + "summary": iteration["implementation"].get("summary"), + "candidate": iteration["candidate"], + }) + elif attempt.get("role") == "reviewer": packet = packets_by_attempt.get(attempt["id"]) if packet is not None: attempts.append({ @@ -231,8 +246,10 @@ def fail_budget_exhausted( try: run = store.run(run_id) snapshot = self._review_snapshot(run) - if snapshot.get("task", {}).get("workflow") != "branch-review": - raise ConflictError("queued budget failure is not a branch review") + if snapshot.get("task", {}).get("workflow") not in { + "branch-review", "issue-delivery", + }: + raise ConflictError("queued budget failure is not a review workflow") error = { "error": "BUDGET_EXHAUSTED", "message": "run wall-time budget is exhausted", @@ -487,9 +504,25 @@ def _resolve_snapshot( raise CapabilityUnavailable( "live issue-delivery implementer execution is not available yet" ) - if (not isinstance(internal_implementation_fixture, dict) - or set(internal_implementation_fixture) - != {"writes", "delay_seconds"}): + valid_fixture = ( + isinstance(internal_implementation_fixture, dict) + and set(internal_implementation_fixture) + == {"writes", "delay_seconds"} + ) + if (isinstance(internal_implementation_fixture, dict) + and set(internal_implementation_fixture) == {"iterations"}): + fixtures = internal_implementation_fixture["iterations"] + valid_fixture = ( + isinstance(fixtures, list) + and 1 <= len(fixtures) + <= task["budget"]["max_revisions"] + 1 + and all( + isinstance(item, dict) + and set(item) == {"writes", "delay_seconds"} + for item in fixtures + ) + ) + if not valid_fixture: raise ContractError("internal implementation fixture is invalid") snapshot["internal_implementation_fixture"] = json.loads( canonical_json(internal_implementation_fixture) @@ -549,10 +582,14 @@ def _continue_preparation( store.validate_predecessor(run_id, fencing_token, supersedes_run_id) validated_supersedes_run_id = supersedes_run_id validate_task(task, require_existing_repo=True) + minimum_headless_invocations = ( + 3 if task["workflow"] == "issue-delivery" else 2 + ) if (task["lead"]["mode"] == "headless" - and task["budget"]["max_worker_invocations"] < 2): + and task["budget"]["max_worker_invocations"] + < minimum_headless_invocations): raise ContractError( - "headless branch review requires at least two worker invocations" + "headless workflow has insufficient worker invocations" ) worktree = store.preparation_worktree( run_id, fencing_token, Path(task["project"]["repo_path"]), @@ -785,7 +822,8 @@ def cancel(self, run_id: str) -> dict[str, Any]: try: snapshot = self._review_snapshot(run) handoff = store.handoff_snapshot(run_id) - if (snapshot.get("task", {}).get("workflow") == "branch-review" + if (snapshot.get("task", {}).get("workflow") + in {"branch-review", "issue-delivery"} and handoff is not None): terminal_artifacts = self._paused_review_terminal_artifacts( store, @@ -826,11 +864,12 @@ def handoff_claim( run = store.run(run_id) handoff_before_claim = store.handoff_snapshot(run_id) if (handoff_before_claim is not None - and handoff_before_claim.packet.get("workflow") == "branch-review" + and handoff_before_claim.packet.get("workflow") + in {"branch-review", "issue-delivery"} and self._review_snapshot(run)["task"]["lead"]["mode"] == "headless"): raise ConflictError( - "headless branch review does not accept a host claim" + "headless review does not accept a host claim" ) claim = store.claim_handoff(run_id, expected_version, owner, decoded) snapshot = store.handoff_snapshot(run_id) @@ -867,7 +906,7 @@ def _review_gate( snapshot: dict[str, Any], decision: dict[str, Any], ) -> tuple[dict[str, Any], dict[str, Any]]: - packet = validate_branch_review_handoff(handoff.packet, snapshot) + packet = validate_saved_review_handoff(handoff.packet, snapshot) validate_handoff_decision_evidence(decision, packet) revisions_used = sum( entry["decision"]["disposition"] == "revise" @@ -1059,11 +1098,15 @@ def _continue_branch_review_submission( ) if gate["action"] == "repeat_review": decision = entry["decision"] - requeue = store.requeue_review_revision( - run_id, - handoff.handoff_id, - decision["submission_id"], - decision["submission_hash"], + workflow = snapshot["task"]["workflow"] + requeue_method = ( + store.requeue_delivery_revision + if workflow == "issue-delivery" + else store.requeue_review_revision + ) + requeue = requeue_method( + run_id, handoff.handoff_id, + decision["submission_id"], decision["submission_hash"], ) if requeue["action"] == "requeued": queued = store.run(run_id) @@ -1091,7 +1134,7 @@ def _continue_branch_review_submission( } error = { "error": "BUDGET_EXHAUSTED", - "message": "branch review worker invocation budget is exhausted", + "message": f"{workflow} worker invocation or wall budget is exhausted", } elif gate["action"] == "budget_exhausted": error = { @@ -1312,8 +1355,9 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st if run["state"] in TERMINAL_STATES: raise ConflictError("terminal run cannot resume; start a superseding run") if run["state"] == "awaiting_host" and run["phase"] is None: handoff = store.handoff_snapshot(run_id) - if handoff is None or handoff.packet.get("workflow") != "branch-review": - raise ConflictError("run has no resumable branch review handoff") + if (handoff is None or handoff.packet.get("workflow") + not in {"branch-review", "issue-delivery"}): + raise ConflictError("run has no resumable review handoff") snapshot = self._review_snapshot(run) if snapshot["task"]["lead"]["mode"] == "headless": continuation = self._continue_headless_lead( @@ -1332,12 +1376,13 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st raise ConflictError("host-led handoff must be completed by its host") elif run["state"] == "awaiting_host" and run["phase"] == "handoff_submitted": handoff = store.handoff_snapshot(run_id) - if handoff is None or handoff.packet.get("workflow") != "branch-review": - raise ConflictError("run has no resumable branch review submission") + if (handoff is None or handoff.packet.get("workflow") + not in {"branch-review", "issue-delivery"}): + raise ConflictError("run has no resumable review submission") snapshot = self._review_snapshot(run) entry = store.recorded_handoff_submission(run_id, handoff.handoff_id) if entry is None: - raise ConflictError("recorded branch review submission is missing") + raise ConflictError("recorded review submission is missing") continuation = self._continue_branch_review_submission( store, run_id, handoff, snapshot, entry, ) diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 01ee311..1cb6d48 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -2547,6 +2547,163 @@ def requeue_review_revision( self.connection.execute("ROLLBACK") raise + def requeue_delivery_revision( + self, + run_id: str, + handoff_id: str, + submission_id: str, + submission_hash: str, + ) -> dict[str, Any]: + """Consume a delivery revise decision and atomically return to its writer.""" + if not all( + isinstance(value, str) and value + for value in (handoff_id, submission_id, submission_hash) + ): + raise ContractError("delivery revision identifiers are invalid") + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT r.state,r.phase,r.version,r.mutable_snapshot,h.status,h.sequence," + "h.packet_json,s.disposition,s.decision_json " + "FROM runs r JOIN handoffs h ON h.run_id=r.id " + "JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE r.id=? AND h.id=? AND s.submission_id=? " + "AND s.submission_hash=? AND s.outcome='recorded'", + (run_id, handoff_id, submission_id, submission_hash), + ).fetchone() + if not row: + raise ConflictError("recorded delivery revision is missing") + if row["disposition"] != "revise": + raise ConflictError("delivery submission is not a revision request") + latest_sequence = self.connection.execute( + "SELECT MAX(sequence) FROM handoffs WHERE run_id=?", (run_id,), + ).fetchone()[0] + if row["status"] == "consumed": + action = ( + "requeued" + if row["sequence"] == latest_sequence + and row["state"] == "queued" + and row["phase"] is None + else "already_advanced" + ) + self.connection.execute("COMMIT") + return { + "action": action, + "version": row["version"], + "replayed": True, + } + if (row["state"] != "awaiting_host" + or row["phase"] != "handoff_submitted" + or row["status"] != "submitted" + or row["sequence"] != latest_sequence): + raise ConflictError("delivery revision handoff is no longer current") + try: + snapshot = json.loads(row["mutable_snapshot"]) + packet = json.loads(row["packet_json"]) + decision = json.loads(row["decision_json"]) + task = snapshot["task"] + budget = task["budget"] + candidate = snapshot["candidate"] + delivery = snapshot["delivery_workspace"] + except (KeyError, TypeError, json.JSONDecodeError) as exc: + raise ConflictError("frozen delivery revision is invalid") from exc + if (not isinstance(snapshot, dict) + or canonical_json(snapshot) != row["mutable_snapshot"] + or task.get("workflow") != "issue-delivery" + or packet.get("workflow") != "issue-delivery" + or packet.get("candidate_sha256") + != candidate.get("candidate_sha256") + or decision.get("submission_id") != submission_id + or decision.get("submission_hash") != submission_hash): + raise ConflictError("frozen delivery revision changed its evidence") + max_revisions = budget.get("max_revisions") + max_invocations = budget.get("max_worker_invocations") + lead_mode = task.get("lead", {}).get("mode") + if (type(max_revisions) is not int or max_revisions < 0 + or type(max_invocations) is not int or max_invocations < 1 + or lead_mode not in {"host", "headless"}): + raise ConflictError("frozen delivery budget is invalid") + revisions = self.connection.execute( + "SELECT COUNT(*) FROM handoff_submissions s " + "JOIN handoffs h ON h.id=s.handoff_id " + "WHERE h.run_id=? AND s.outcome='recorded' " + "AND s.disposition='revise' AND h.sequence<=?", + (run_id, row["sequence"]), + ).fetchone()[0] + invocations = self.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()[0] + required_invocations = 3 if lead_mode == "headless" else 2 + wall_exhausted = self.remaining_wall_seconds(run_id) == 0 + if (revisions > max_revisions + or invocations + required_invocations > max_invocations + or wall_exhausted): + self.connection.execute("COMMIT") + return { + "action": "budget_exhausted", + "version": row["version"], + "replayed": False, + "revisions_requested": revisions, + "worker_invocations": invocations, + "wall_exhausted": wall_exhausted, + } + new_snapshot = json.loads(canonical_json(snapshot)) + review_fixture = new_snapshot.pop("internal_review_fixture", None) + if review_fixture is not None: + new_snapshot["pending_review_fixture"] = { + field: review_fixture[field] + for field in ("verdict", "summary", "findings") + } + revision_request = { + "handoff_id": handoff_id, + "sequence": row["sequence"], + "submission_id": submission_id, + "submission_hash": submission_hash, + "previous_candidate_sha256": candidate["candidate_sha256"], + "reason": decision["reason"], + "review": packet["review"], + "checks": packet["checks"], + "evidence_refs": decision["evidence_refs"], + } + new_snapshot["revision_request"] = revision_request + for field in ("candidate", "workspace", "check_workspace"): + new_snapshot.pop(field, None) + encoded_snapshot = canonical_json(new_snapshot) + delivery_path = str(Path(delivery["path"]).resolve(strict=True)) + now, version = _utc_now(), row["version"] + 1 + self.connection.execute( + "UPDATE handoffs SET status='consumed',closed_at=? WHERE id=?", + (now, handoff_id), + ) + self.connection.execute( + "UPDATE runs SET mutable_snapshot=?,worktree_path=?,state='queued'," + "phase=NULL,version=?,updated_at=? WHERE id=?", + (encoded_snapshot, delivery_path, version, now, run_id), + ) + payload = canonical_json({ + "handoff_id": handoff_id, + "submission_id": submission_id, + "previous_candidate_sha256": candidate["candidate_sha256"], + "revisions_requested": revisions, + "worker_invocations": invocations, + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'delivery.revision_queued',?,?)", + (run_id, version, payload, now), + ) + self.connection.execute("COMMIT") + return { + "action": "requeued", + "version": version, + "replayed": False, + "revisions_requested": revisions, + "worker_invocations": invocations, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + def complete_handoff_terminal( self, run_id: str, diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 5ab120d..d8dc03f 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -405,12 +405,19 @@ def _commit_delivery_candidate( delivery = snapshot["delivery_workspace"] source_repo = Path(task["project"]["repo_path"]).resolve(strict=True) workspace = Path(delivery["path"]).resolve(strict=True) + iterations = list(snapshot.get("delivery_iterations", [])) + iteration = len(iterations) + 1 + parent_oid = ( + iterations[-1]["candidate"]["commit_oid"] if iterations + else delivery["baseline_oid"] + ) candidate, patch = freeze_delivery_candidate( source_repo, workspace, delivery["baseline_oid"], task["scope"]["write_paths"], run_id, + parent_oid=parent_oid, ) project_id = self.store.run(run_id)["project_id"] scope_paths = tuple(dict.fromkeys( @@ -430,6 +437,10 @@ def _commit_delivery_candidate( scope_paths, required_clean_paths=config_paths, candidate_sha256=candidate["candidate_sha256"], + workspace_name=( + "review-worktree" if iteration == 1 + else f"review-worktree-{iteration}" + ), ) check_workspace = prepare_check_workspace( source_repo, @@ -439,8 +450,11 @@ def _commit_delivery_candidate( candidate["commit_oid"], scope_paths, required_clean_paths=config_paths, + workspace_name=( + "check-worktree" if iteration == 1 + else f"check-worktree-{iteration}" + ), ) - iteration = len(snapshot.get("delivery_iterations", [])) + 1 candidate_record = { **candidate, "iteration": iteration, @@ -466,12 +480,17 @@ def _commit_delivery_candidate( new_snapshot["internal_review_fixture"] = validate_review_document( fixture_document, task, review_workspace, ) - iterations = list(new_snapshot.get("delivery_iterations", [])) - iterations.append({ + revision_request = new_snapshot.pop("revision_request", None) + iteration_record = { "iteration": iteration, "candidate": candidate_record, "implementation": evidence, - }) + "workspace": review_workspace, + "check_workspace": check_workspace, + } + if revision_request is not None: + iteration_record["revision_request"] = revision_request + iterations.append(iteration_record) new_snapshot["delivery_iterations"] = iterations artifacts = list(stream_artifacts) @@ -589,7 +608,8 @@ def import_durable(self, run_id: str) -> str: semantic_error = str(exc) receipt["error"] = "IMPLEMENTATION_OUTPUT_INVALID" receipt["message"] = semantic_error - elif (role == "lead" and workflow_review and not receipt["cancelled"] + elif (role == "lead" and (workflow_review or workflow_delivery) + and not receipt["cancelled"] and not receipt["timed_out"] and receipt["returncode"]==0): try: handoff = self.store.handoff_snapshot(run_id) @@ -636,7 +656,7 @@ def import_durable(self, run_id: str) -> str: if role == "lead" else "WORKFLOW_OUTPUT_INVALID" ) payload["message"]=semantic_error - if workflow_review: + if workflow_review or workflow_delivery: from .service import Service prior_attempts, prior_dispositions = Service._saved_review_progress( self.store, @@ -649,7 +669,7 @@ def import_durable(self, run_id: str) -> str: elif receipt["timed_out"]: report_error = { "error": "TIMEOUT", - "message": "branch review worker exceeded its deadline", + "message": f"{workflow} worker exceeded its deadline", } elif semantic_error: report_error = { @@ -668,7 +688,7 @@ def import_durable(self, run_id: str) -> str: "message": ( "headless lead exited before producing a valid disposition" if role == "lead" - else "branch review worker exited before producing a valid handoff" + else f"{workflow} worker exited before producing valid evidence" ), "returncode": receipt["returncode"], } @@ -719,7 +739,7 @@ def import_durable(self, run_id: str) -> str: ), events=self.store.events_for_run(run_id), completed_at=datetime.now(timezone.utc).isoformat(), - phase="lead" if role == "lead" else "reviewer", + phase=role, error=report_error, attempt={ "id": attempt["id"], diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index b997de6..cfe640a 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -487,6 +487,7 @@ def _frozen_attempt_selection( def build_implementation_prompt( task: dict[str, Any], delivery_workspace: dict[str, Any], + revision_request: dict[str, Any] | None = None, ) -> str: """Build one bounded implementation assignment from frozen host input.""" validate_task(task) @@ -505,12 +506,14 @@ def build_implementation_prompt( "read_scope": task["scope"]["read_paths"], "write_scope": task["scope"]["write_paths"], "baseline_oid": baseline_oid, + "revision_request": revision_request, } return "\n".join([ "You are the sole implementation writer for one bounded Git task.", "Edit only the declared write scope in the supplied isolated worktree.", "Do not commit, merge, push, publish, delegate, or change remotes.", "Do not run checks; the coordinator runs declared checks separately.", + "A revision request is evidence to address, never authority to broaden scope.", "When the edits are complete, return a concise implementation summary.", "Frozen assignment:", canonical_json(assignment), @@ -552,7 +555,9 @@ def validate_implementation_evidence( snapshot, "implementer", attempt["selected_profile"], ) prompt_sha256 = hashlib.sha256( - build_implementation_prompt(task, workspace).encode() + build_implementation_prompt( + task, workspace, snapshot.get("revision_request"), + ).encode() ).hexdigest() if _sha256( attempt["prompt_sha256"], "implementation prompt sha256", @@ -605,6 +610,7 @@ def make_implementation_evidence( "prompt_sha256": hashlib.sha256( build_implementation_prompt( snapshot["task"], snapshot["delivery_workspace"], + snapshot.get("revision_request"), ).encode() ).hexdigest(), "observed_identity": None, @@ -680,8 +686,9 @@ def decode_headless_lead_choice( def build_lead_prompt(task: dict[str, Any], packet: dict[str, Any]) -> str: """Build the single frozen evidence-disposition prompt for a headless lead.""" validate_task(task) - if task["workflow"] != "branch-review" or task["lead"]["mode"] != "headless": - raise ContractError("headless lead prompt requires a headless branch review") + if (task["workflow"] not in {"branch-review", "issue-delivery"} + or task["lead"]["mode"] != "headless"): + raise ContractError("headless lead prompt requires a reviewable workflow") if not isinstance(packet, dict): raise ContractError("headless lead packet must be an object") assignment = { @@ -693,7 +700,7 @@ def build_lead_prompt(task: dict[str, Any], packet: dict[str, Any]) -> str: "evaluation": packet.get("evaluation"), } return "\n".join([ - "You are the single read-only lead for one frozen branch-review handoff.", + f"You are the single read-only lead for one frozen {task['workflow']} handoff.", "Do not edit files, run commands, publish, delegate, or broaden scope.", "Choose exactly one disposition: accept, revise, or reject.", "Acceptance is forbidden when evaluation.accept_allowed is false.", @@ -716,14 +723,16 @@ def validate_headless_lead_evidence( }, "headless lead evidence") if document["schema_version"] != 1 or type(document["schema_version"]) is not int: raise ContractError("headless lead evidence schema_version is invalid") - if document["workflow"] != "branch-review": - raise ContractError("headless lead evidence workflow is invalid") if not isinstance(snapshot, dict) or not isinstance(handoff, dict): raise ContractError("headless lead frozen inputs are invalid") task = snapshot.get("task") packet = handoff.get("packet") if not isinstance(task, dict) or not isinstance(packet, dict): raise ContractError("headless lead frozen inputs are incomplete") + if (task.get("workflow") not in {"branch-review", "issue-delivery"} + or document["workflow"] != task["workflow"] + or packet.get("workflow") != task["workflow"]): + raise ContractError("headless lead evidence workflow is invalid") if task.get("lead", {}).get("mode") != "headless": raise ContractError("headless lead evidence requires headless mode") handoff_id = _text(document["handoff_id"], "headless lead handoff id", maximum=200) @@ -817,7 +826,7 @@ def make_headless_lead_evidence( normalized_choice = validate_headless_lead_choice(choice, packet) document = { "schema_version": 1, - "workflow": "branch-review", + "workflow": snapshot["task"]["workflow"], "candidate_sha256": normalized_choice["candidate_sha256"], "handoff_id": handoff["handoff_id"], "packet_sha256": handoff["packet_sha256"], diff --git a/plugin/core/src/devsquad/workspaces.py b/plugin/core/src/devsquad/workspaces.py index 37e14c8..9bc4051 100644 --- a/plugin/core/src/devsquad/workspaces.py +++ b/plugin/core/src/devsquad/workspaces.py @@ -212,8 +212,11 @@ def reset_check_workspace( """Reset only the exact run-owned check worktree before another check pass.""" review = review_workspace.resolve(strict=True) checks = check_workspace.resolve(strict=True) - if (review.name != "review-worktree" - or checks != review.parent / "check-worktree"): + review_prefix = "review-worktree" + if not review.name.startswith(review_prefix): + raise ContractError("check workspace is not the review run's owned sibling") + suffix = review.name[len(review_prefix):] + if checks != review.parent / f"check-worktree{suffix}": raise ContractError("check workspace is not the review run's owned sibling") _validate_workspace(review, review, target_oid, scope_paths) _validate_workspace( @@ -237,6 +240,7 @@ def _prepare_detached_workspace( scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) project = _validate_segment(project_id, "project id") run = _validate_segment(run_id, "run id") + name = _validate_segment(name, "workspace name") workspace = ( runtime.resolve() / "projects" / project / "runs" / run / name ) @@ -273,13 +277,14 @@ def prepare_review_workspace( *, required_clean_paths: Iterable[str] = (), candidate_sha256: str | None = None, + workspace_name: str = "review-worktree", ) -> dict[str, object]: """Create or validate one detached, run-owned worktree at the target commit.""" repo = source_repo.resolve(strict=True) scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) assert_clean_inputs(repo, scopes, required_clean_paths) workspace, scopes = _prepare_detached_workspace( - repo, runtime, project_id, run_id, target_oid, scopes, "review-worktree", + repo, runtime, project_id, run_id, target_oid, scopes, workspace_name, ) changed = _decode_paths( _git( @@ -318,13 +323,14 @@ def prepare_check_workspace( scope_paths: Iterable[str], *, required_clean_paths: Iterable[str] = (), + workspace_name: str = "check-worktree", ) -> dict[str, object]: """Create an independent candidate worktree for trusted declared checks.""" repo = source_repo.resolve(strict=True) scopes = tuple(_normalized_relative(path, "scope path") for path in scope_paths) assert_clean_inputs(repo, scopes, required_clean_paths) workspace, scopes = _prepare_detached_workspace( - repo, runtime, project_id, run_id, target_oid, scopes, "check-worktree", + repo, runtime, project_id, run_id, target_oid, scopes, workspace_name, ) return { "schema_version": 1, @@ -465,6 +471,7 @@ def _freeze_delivery_candidate_unlocked( baseline_oid: str, write_paths: Iterable[str], run_id: str, + parent_oid: str | None = None, ) -> tuple[dict[str, object], bytes]: """Commit one scoped candidate locally and return its stable patch identity.""" repo = source_repo.resolve(strict=True) @@ -472,6 +479,8 @@ def _freeze_delivery_candidate_unlocked( run = _validate_segment(run_id, "run id") if delivery.name != "delivery-worktree": raise ContractError("delivery workspace is not a run-owned delivery worktree") + expected_parent = parent_oid or baseline_oid + expected_parent = resolve_commit(delivery, expected_parent) head_oid = resolve_commit(delivery, "HEAD") _validate_workspace( repo, @@ -481,14 +490,14 @@ def _freeze_delivery_candidate_unlocked( require_clean=False, ) dirties = dirty_paths(delivery) - if head_oid != baseline_oid: + if head_oid != expected_parent: if dirties: raise ContractError("frozen delivery candidate has later workspace changes") parents = _git( delivery, "rev-list", "--parents", "-n", "1", head_oid, ).decode().strip().split() - if parents != [head_oid, baseline_oid]: - raise ContractError("delivery candidate is not a single local baseline commit") + if parents != [head_oid, expected_parent]: + raise ContractError("delivery candidate is not a single local parent commit") marker = _git( delivery, "show", "-s", "--format=%s%x00%ae", head_oid, ).decode("utf-8", "strict").rstrip("\n").split("\0") @@ -533,7 +542,11 @@ def _freeze_delivery_candidate_unlocked( snapshot, patch = _candidate_snapshot( delivery, baseline_oid, commit_oid, write_paths, ) - if patch != staged_patch: + committed_delta = _git( + delivery, + "diff", "--binary", "--no-ext-diff", expected_parent, commit_oid, "--", + ) + if committed_delta != staged_patch: raise ContractError("committed delivery patch differs from the staged candidate") if dirty_paths(delivery): raise ContractError("delivery workspace remained dirty after candidate commit") @@ -546,6 +559,8 @@ def freeze_delivery_candidate( baseline_oid: str, write_paths: Iterable[str], run_id: str, + *, + parent_oid: str | None = None, ) -> tuple[dict[str, object], bytes]: """Serialize candidate freezing so competing recovery importers replay it.""" delivery = workspace.resolve(strict=True) @@ -554,7 +569,7 @@ def freeze_delivery_candidate( try: fcntl.flock(descriptor, fcntl.LOCK_EX) return _freeze_delivery_candidate_unlocked( - source_repo, delivery, baseline_oid, write_paths, run_id, + source_repo, delivery, baseline_oid, write_paths, run_id, parent_oid, ) finally: fcntl.flock(descriptor, fcntl.LOCK_UN) diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py index 652f590..a6b2d52 100644 --- a/test/core/test_delivery_workflow.py +++ b/test/core/test_delivery_workflow.py @@ -14,12 +14,13 @@ from devsquad.contracts import ContractError from devsquad.service import Service -from devsquad.store import Store, request_hash +from devsquad.store import ConflictError, Store, request_hash from devsquad.workspaces import ( freeze_delivery_candidate, prepare_delivery_workspace, resolve_commit, ) +from devsquad.workflows import validate_branch_review_evidence class DeliveryWorkspaceTest(unittest.TestCase): @@ -97,6 +98,12 @@ def profile( model="fixture-review-model", permission="read_only", ), + profile( + "fixture-lead", + family="fixture-family-c", + model="fixture-lead-model", + permission="read_only", + ), ], "bindings": {}, } @@ -111,6 +118,7 @@ def profile( "reviewer": [ {"kind": "profile", "id": "fixture-reviewer"} ], + "lead": [{"kind": "profile", "id": "fixture-lead"}], }, "task_classes": {"fixture-delivery-small": "proven"}, "require_different_model_for_review": True, @@ -126,6 +134,11 @@ def profile( "max_concurrency": 1, "unknown_capacity_policy": "allow_bounded", }, + "fixture-lead-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, }, "experiment_budget": {}, } @@ -215,6 +228,25 @@ def implementation_fixture(delay: float = 0) -> dict[str, object]: "delay_seconds": delay, } + @staticmethod + def repair_fixtures() -> dict[str, object]: + return { + "iterations": [ + { + "writes": [ + {"path": "src/app.py", "content": "VALUE = 'wrong'\n"}, + ], + "delay_seconds": 0, + }, + { + "writes": [ + {"path": "src/app.py", "content": "VALUE = 'fixed'\n"}, + ], + "delay_seconds": 0, + }, + ], + } + @staticmethod def clean_review_fixture() -> dict[str, object]: return { @@ -267,6 +299,15 @@ def wait_for_handoff(self, service: Service, run_id: str) -> dict[str, object]: time.sleep(0.05) self.fail(f"delivery handoff did not become ready: {service.status(run_id)}") + def wait_for_terminal(self, service: Service, run_id: str) -> dict[str, object]: + deadline = time.monotonic() + 15 + while time.monotonic() < deadline: + status = service.status(run_id) + if status["state"] in {"succeeded", "failed", "cancelled"}: + return status + time.sleep(0.05) + self.fail(f"delivery run did not terminalize: {service.status(run_id)}") + def assert_source_unchanged(self) -> None: self.assertEqual(resolve_commit(self.repo, "HEAD"), self.baseline) self.assertEqual(self.git(self.repo, "status", "--porcelain"), self.source_status) @@ -538,6 +579,200 @@ def test_required_failure_blocks_delivery_accept_and_allows_reject(self): self.assertEqual(receipt["evaluation"]["required_failures"], ["fixture-check"]) self.assert_source_unchanged() + def test_revision_returns_to_implementer_and_replaces_candidate_evidence(self): + task = self.delivery_task() + task["checks"][0] = { + **task["checks"][0], + "argv": [ + "python3", "-c", + "from pathlib import Path; " + "assert Path('src/app.py').read_text() == \"VALUE = 'fixed'\\n\"", + ], + } + service = Service(self.runtime) + started = service.start( + task, + "repaired-delivery", + _internal_implementation_fixture=self.repair_fixtures(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + first_wait = self.wait_for_handoff(service, started["run_id"]) + first_claim = service.handoff_claim( + started["run_id"], first_wait["version"], "fixture-host-one", + ) + first_packet = first_claim["handoff"]["packet"] + self.assertFalse(first_packet["evaluation"]["accept_allowed"]) + revised = service.handoff_complete( + started["run_id"], first_claim["claim"], + self.decision( + first_packet, "repair-delivery", "revise", "Fix the required value.", + ), + ) + self.assertEqual(revised["continuation"]["action"], "requeued") + self.assertTrue(revised["launched"]) + self.wait_for_candidate(service, started["run_id"]) + + store = Store(service.database, service.artifacts) + try: + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + self.assertEqual(len(snapshot["delivery_iterations"]), 2) + candidates = [ + item["candidate"] for item in snapshot["delivery_iterations"] + ] + self.assertNotEqual( + candidates[0]["candidate_sha256"], candidates[1]["candidate_sha256"], + ) + self.assertNotEqual(candidates[0]["commit_oid"], candidates[1]["commit_oid"]) + self.assertEqual( + candidates[1]["baseline_oid"], candidates[0]["baseline_oid"], + ) + self.assertEqual( + snapshot["delivery_iterations"][1]["revision_request"][ + "previous_candidate_sha256" + ], + candidates[0]["candidate_sha256"], + ) + self.assertEqual( + [item["role"] for item in store.attempts_for_run(started["run_id"])], + ["implementer", "reviewer", "implementer"], + ) + stale_evidence = { + field: first_packet[field] + for field in ( + "schema_version", "workflow", "candidate_sha256", "base_oid", + "target_oid", "review", "checks", "evaluation", "attempt", + ) + } + with self.assertRaisesRegex( + ContractError, "changes frozen candidate_sha256", + ): + validate_branch_review_evidence(stale_evidence, snapshot) + finally: + store.close() + + service.resume(started["run_id"]) + second_wait = self.wait_for_handoff(service, started["run_id"]) + second_claim = service.handoff_claim( + started["run_id"], second_wait["version"], "fixture-host-two", + ) + second_packet = second_claim["handoff"]["packet"] + self.assertNotEqual( + first_packet["candidate_sha256"], second_packet["candidate_sha256"], + ) + self.assertTrue(second_packet["evaluation"]["accept_allowed"]) + completed = service.handoff_complete( + started["run_id"], second_claim["claim"], + self.decision( + second_packet, "accept-repair", "accept", "Repair accepted.", + ), + ) + self.assertEqual(completed["state"], "succeeded") + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual( + [item["role"] for item in receipt["attempts"]], + ["implementer", "implementer", "reviewer", "reviewer"], + ) + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], + ["revise", "accept"], + ) + self.assertEqual(receipt["revisions"]["executed"], 1) + with self.assertRaises(ConflictError): + service.handoff_complete( + started["run_id"], first_claim["claim"], + self.decision( + first_packet, "stale-reject", "reject", "Stale evidence.", + ), + ) + self.assert_source_unchanged() + + def test_delivery_revision_and_invocation_budgets_fail_before_new_writer(self): + for suffix, max_revisions, max_invocations in ( + ("revision", 0, 5), + ("invocation", 1, 3), + ): + with self.subTest(budget=suffix): + task = self.delivery_task() + task["budget"]["max_revisions"] = max_revisions + task["budget"]["max_worker_invocations"] = max_invocations + service = Service(self.runtime / suffix) + started = service.start( + task, + f"exhausted-{suffix}", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], f"host-{suffix}", + ) + packet = claimed["handoff"]["packet"] + completed = service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision( + packet, f"revise-{suffix}", "revise", "Request repair.", + ), + ) + self.assertEqual(completed["state"], "failed") + self.assertFalse(completed["launched"]) + store = Store(service.database, service.artifacts) + try: + self.assertEqual( + [item["role"] for item in store.attempts_for_run( + started["run_id"] + )], + ["implementer", "reviewer"], + ) + finally: + store.close() + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["error"]["error"], "BUDGET_EXHAUSTED") + self.assertEqual(receipt["lead"]["disposition"], "revise") + + def test_headless_delivery_lead_accepts_the_candidate(self): + task = self.delivery_task() + task["lead"] = {"mode": "headless"} + service = Service(self.runtime) + started = service.start( + task, + "headless-delivery", + _internal_implementation_fixture=self.implementation_fixture(), + _internal_review_fixture=self.clean_review_fixture(), + _internal_lead_fixture={ + "disposition": "accept", + "reason": "All required evidence passes.", + }, + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + terminal = self.wait_for_terminal(service, started["run_id"]) + self.assertEqual(terminal["state"], "succeeded") + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["lead"]["mode"], "headless") + self.assertEqual(receipt["lead"]["disposition"], "accept") + self.assertEqual( + [item["role"] for item in receipt["attempts"]], + ["implementer", "reviewer"], + ) + self.assertEqual(len(receipt["lead"]["attempts"]), 1) + self.assertEqual(receipt["accounting"]["worker_invocations"], 3) + def test_live_implementer_cannot_be_resumed_into_a_second_writer(self): service = Service(self.runtime) started = service.start( @@ -591,8 +826,57 @@ def test_durable_out_of_scope_implementation_cannot_publish_a_candidate(self): self.assertNotIn( "candidate-1.json", {item["name"] for item in result["artifacts"]}, ) + self.assertTrue({ + "receipt.json", "receipt.md", "events.jsonl", + "artifact-manifest.json", "result-receipt.json", + } <= {item["name"] for item in result["artifacts"]}) self.assert_source_unchanged() + def test_revision_resume_after_prelaunch_crash_creates_one_repair_writer(self): + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "revision-prelaunch-crash", + _internal_implementation_fixture=self.repair_fixtures(), + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], "fixture-crash-host", + ) + packet = claimed["handoff"]["packet"] + original_spawn = service._spawn_daemon + service._spawn_daemon = lambda *args, **kwargs: 0 + try: + saved = service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision( + packet, "repair-after-crash", "revise", "Repair candidate.", + ), + ) + finally: + service._spawn_daemon = original_spawn + self.assertTrue(saved["launched"]) + self.assertEqual(service.status(started["run_id"])["state"], "queued") + + resumed = Service(self.runtime).resume(started["run_id"]) + self.assertTrue(resumed["launched"]) + self.wait_for_candidate(service, started["run_id"]) + store = Store(service.database, service.artifacts) + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual( + [item["role"] for item in attempts], + ["implementer", "reviewer", "implementer"], + ) + self.assertEqual(len({item["id"] for item in attempts}), 3) + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + self.assertEqual(len(snapshot["delivery_iterations"]), 2) + finally: + store.close() + if __name__ == "__main__": unittest.main() From 7cad26ea851b381fd9f2b8b852a6a2021a0bcd68 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sat, 26 Sep 2026 23:59:22 -0700 Subject: [PATCH 091/197] docs: checkpoint bounded delivery revisions --- docs/plans/engineering-team/M5-STATUS.md | 35 ++++++++++++++++++++---- docs/plans/engineering-team/RESUME.md | 25 +++++++++++++---- docs/plans/engineering-team/backlog.json | 9 ++++++ 3 files changed, 58 insertions(+), 11 deletions(-) diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index 1af7be9..a15ef79 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -11,11 +11,11 @@ two-harness gate passes. | Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | verified offline at `0e88d73` | | Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `0e88d73` | | Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | router verified; workflow pending | -| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | first-candidate review/check/handoff verified offline at `30b98df`; revision/stale-evidence workflow gate pending | -| Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | pending | -| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | pending | -| Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | pending | -| Crash recovery | Killing a live implementation supervisor cannot create a duplicate writer on resume | pending | +| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | verified offline through two distinct candidates at `4c76887` | +| Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | correction and revision/invocation budgets verified offline at `4c76887`; delivery fallback faults pending | +| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | mandatory-check gate and accept/reject receipts verified offline at `a1199c6`; invalid-output fault matrix pending | +| Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | successful repair and headless history verified offline at `4c76887`; failed fallback/cancellation history pending | +| Crash recovery | Killing a live implementation supervisor cannot create a duplicate writer on resume | initial live-writer and revised prelaunch recovery verified offline at `4c76887`; live revised-writer fault gate pending | | Live acceptance | One bounded issue completes across at least two authenticated subscription harnesses with different verified models | pending | ## Boundary @@ -90,3 +90,28 @@ The complete core discovery is **226 tests, suite OK with 2 optional-SDK skips**, with `ResourceWarning` promoted to error; **220 Bash assertions** passed. The next implementation slice is Plan 07-02 Task 3, retaining Task 2's unproven live/stale-revision requirements rather than marking M5 complete. + +## Plan 07-02 checkpoint 2 + +At `a1199c6`, host accept/reject applies the trusted gate to `issue-delivery`, +blocks acceptance after a required-check failure, and emits the complete +five-report terminal set with implementation plus review evidence. + +At `4c76887`, `revise` atomically consumes the saved handoff, switches the +writer fence back to the delivery worktree, binds prior review/check evidence +into a frozen revision request and launches a new implementer. The next local +commit is parented by the prior candidate while its identity and patch still +describe the complete baseline-to-candidate change. Each iteration gets +separate review/check worktrees, so stale live evidence cannot validate +against a replacement candidate; terminal history validates each archived +candidate against its own saved workspace. + +The seeded repair passes implementation → failed mandatory check → host revise +→ new implementation → new review/check → accept with distinct candidate +hashes and roles `[implementer, reviewer, implementer, reviewer]`. Revision and +invocation exhaustion fail before another writer launches, a saved prelaunch +revision resumes to exactly one repair writer, and a headless delivery lead +terminalizes through its own fenced attempt. The complete gate is **237 core +tests (2 optional-SDK skips)** with `ResourceWarning` promoted to error and +**220 Bash assertions**. Delivery fallback/failure/cancellation history, live +Claude implementation and the live two-harness proof remain open. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index bce246e..5547ff7 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -109,6 +109,18 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne hash. The complete gate is 226 core tests discovered (suite OK, 2 optional SDK skips) and 220 Bash assertions. This does not complete lead disposition, revise-to-implementation, all terminal reports or the live two-harness gate. +- M5 Plan 07-02 bounded revisions are verified offline at `4c76887`, following + delivery accept/reject receipts at `a1199c6`. A saved `revise` transaction + now returns the writer fence to the delivery worktree, binds prior evidence + into the next prompt, creates a distinct child candidate and new review/check + worktrees, rejects stale live evidence and preserves both candidates plus all + successful worker attempts and dispositions. Required checks remain + non-overridable; revision/invocation exhaustion stops before a new writer; a + simulated prelaunch crash resumes exactly one repair writer; and headless + delivery acceptance uses its own fenced lead attempt. The gate is 237 core + tests (2 optional-SDK skips) and 220 Bash assertions. Delivery fallback, + failure/cancellation history, live Claude execution and the live two-harness + proof remain open. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -172,12 +184,13 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M5 Plan 07-02 Task 3 from `30b98df`: complete delivery lead - disposition and make bounded `revise` return to the fenced implementer; - retain every candidate/attempt/check/disposition in terminal reports. - Exercise seeded repair, stale-candidate rejection, mandatory-check blocking, - fallback/deadline bounds and crash recovery. First-candidate offline - review/check/handoff already works; do not rebuild that slice. +2. Continue M5 Plan 07-02 Task 3 from `4c76887`: complete delivery-specific + implementer/reviewer fallback faults, deadline/cancellation terminal history + and remaining live-revised-writer recovery. Then wire the already-conformed + Claude adapter into live implementation and run the live two-harness gate + only after normal provider login. Accept/reject, seeded repair, stale + evidence, required-check enforcement, finite revision/invocation budgets, + headless lead and prelaunch crash recovery are green; do not rebuild them. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index cc1f3c4..6475663 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -249,6 +249,15 @@ "artifact": "M5-STATUS.md", "recorded_at": "2026-09-26T15:01:21Z", "availability": "tracked_tests" + }, + { + "kind": "implementation_checkpoint", + "revision": "4c76887", + "command_or_action": "237 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions and seeded two-candidate repair/headless/crash regressions", + "outcome": "Issue-delivery lead accept/reject and bounded revise now terminalize with complete candidate-bound reports; revise returns atomically to one fenced implementer, creates distinct candidate review/check workspaces, rejects stale live evidence and enforces revision/invocation budgets. Delivery fallback/failure/cancellation history and live two-harness proof remain open", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-26T23:58:00-07:00", + "availability": "tracked_tests" } ], "blocker": null From b7d90ccaedfe8430d398f36be5be88903dda1045 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 27 Sep 2026 00:09:55 -0700 Subject: [PATCH 092/197] feat: wire live delivery and bounded fallbacks --- .../src/devsquad/claude_delivery_worker.py | 288 ++++++++++++++++++ plugin/core/src/devsquad/delivery_worker.py | 15 +- plugin/core/src/devsquad/detached.py | 8 +- plugin/core/src/devsquad/service.py | 57 +++- plugin/core/src/devsquad/supervisor.py | 16 +- plugin/core/src/devsquad/workflows.py | 13 +- test/core/test_delivery_workflow.py | 175 ++++++++++- 7 files changed, 548 insertions(+), 24 deletions(-) create mode 100644 plugin/core/src/devsquad/claude_delivery_worker.py diff --git a/plugin/core/src/devsquad/claude_delivery_worker.py b/plugin/core/src/devsquad/claude_delivery_worker.py new file mode 100644 index 0000000..8a4eeb2 --- /dev/null +++ b/plugin/core/src/devsquad/claude_delivery_worker.py @@ -0,0 +1,288 @@ +"""Run one frozen Claude CLI implementation inside the durable writer fence.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +from typing import Any + +from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported +from .store import canonical_json +from .workflows import build_implementation_prompt, make_implementation_evidence + + +MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 +MAX_CLAUDE_OUTPUT_BYTES = 2 * 1024 * 1024 +ADAPTER_FIELDS = { + "schema_version", "harness", "transport", "binary", "binary_sha256", + "harness_version", "model_provider", "permission_args", "error_patterns", + "denied_pattern", +} + + +def _manifest_path() -> Path: + source = Path(__file__).resolve().parents[2] / "adapters" / "claude" / "adapter.json" + if source.is_file(): + return source + installed = ( + Path(sys.prefix) / "share" / "devsquad" / "adapters" / "claude" + / "adapter.json" + ) + if installed.is_file(): + return installed + raise CapabilityUnavailable("Claude adapter manifest is unavailable") + + +def freeze_claude_implementer(selected: dict[str, Any]) -> dict[str, Any]: + """Resolve one exact subscription Claude writer during run preflight.""" + from .adapters import ( + AdapterManifest, + DENIED_PATTERN, + ERROR_PATTERNS, + harness_version, + ) + if not isinstance(selected, dict) or not isinstance(selected.get("profile"), dict): + raise ContractError("frozen implementer selection is invalid") + profile = selected["profile"] + if profile.get("harness") != "claude": + raise CapabilityUnavailable( + f"selected implementer harness is not Claude: {profile.get('harness')}" + ) + if profile.get("permission_policy") != "workspace_write": + raise ProfileUnsupported("Claude implementer requires workspace_write") + if set(profile.get("required_tools", [])) - {"read", "write"}: + raise ProfileUnsupported("Claude implementer requests unsupported tools") + effort = profile.get("effort") + if (not isinstance(effort, dict) or effort.get("transport") != "native" + or not isinstance(effort.get("value"), str) or not effort["value"]): + raise ProfileUnsupported("Claude implementer requires an explicit effort") + if not isinstance(profile.get("model_id"), str) or not profile["model_id"]: + raise ProfileUnsupported("Claude implementer requires an exact model id") + + manifest = AdapterManifest.load(_manifest_path()) + binary_name = manifest.resolve_binary() + if not binary_name: + raise CapabilityUnavailable("Claude executable is unavailable") + binary = Path(binary_name).resolve(strict=True) + version = harness_version(str(binary)) + if not version: + raise CapabilityUnavailable("Claude version could not be observed") + if version not in manifest.verified_versions: + raise ProfileUnsupported(f"unverified Claude CLI version: {version}") + return { + "schema_version": 1, + "harness": "claude", + "transport": "cli_exec", + "binary": str(binary), + "binary_sha256": hashlib.sha256(binary.read_bytes()).hexdigest(), + "harness_version": version, + "model_provider": manifest.model_provider or "anthropic", + "permission_args": list(manifest.permission_profiles["workspace_write"]), + "error_patterns": { + code: pattern.pattern for code, pattern in ERROR_PATTERNS + }, + "denied_pattern": DENIED_PATTERN.pattern, + } + + +def _validated(snapshot: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any]]: + adapter = snapshot.get("implementation_adapter") + try: + profile = snapshot["routing"]["roles"]["implementer"]["selected"]["profile"] + except (KeyError, TypeError) as exc: + raise ContractError("frozen Claude implementer selection is missing") from exc + if not isinstance(adapter, dict) or set(adapter) != ADAPTER_FIELDS: + raise ContractError("frozen Claude implementer adapter fields are invalid") + if (adapter["schema_version"] != 1 + or adapter["harness"] != "claude" + or adapter["transport"] != "cli_exec" + or adapter["model_provider"] != "anthropic"): + raise ContractError("frozen Claude implementer adapter identity is invalid") + if (not isinstance(adapter["permission_args"], list) + or adapter["permission_args"] != [ + "--permission-mode", "acceptEdits", "--tools", + "Read,Glob,Grep,Edit,Write", + ] + or not isinstance(adapter["error_patterns"], dict) + or set(adapter["error_patterns"]) != {"AUTH_ERROR", "RATE_LIMITED"} + or not all( + isinstance(value, str) and value + for value in adapter["error_patterns"].values() + ) + or not isinstance(adapter["denied_pattern"], str) + or not adapter["denied_pattern"]): + raise ContractError("frozen Claude implementer policy is invalid") + if (not isinstance(profile, dict) or profile.get("harness") != "claude" + or profile.get("permission_policy") != "workspace_write"): + raise ContractError("frozen profile is not a Claude implementation writer") + return adapter, profile + + +def _structured_result(payload: str) -> tuple[str, str, dict[str, Any]]: + try: + document = json.loads(payload) + except json.JSONDecodeError as exc: + raise ContractError("Claude implementation output is not JSON") from exc + if (not isinstance(document, dict) or document.get("type") != "result" + or document.get("is_error") is True + or not isinstance(document.get("result"), str) + or not document["result"].strip()): + raise ContractError("Claude implementation output has no successful result") + session_id = document.get("session_id") + if not isinstance(session_id, str) or not session_id: + raise ContractError("Claude implementation output has no session id") + raw_usage = document.get("usage") + usage = { + "input_tokens": None, + "output_tokens": None, + "total_tokens": None, + "source": "unavailable", + } + if isinstance(raw_usage, dict): + input_tokens = raw_usage.get("input_tokens") + output_tokens = raw_usage.get("output_tokens") + if all(type(value) is int and value >= 0 for value in ( + input_tokens, output_tokens, + )): + usage = { + "input_tokens": input_tokens, + "output_tokens": output_tokens, + "total_tokens": input_tokens + output_tokens, + "source": "native_reported", + } + return document["result"].strip(), session_id, usage + + +def run(snapshot: dict[str, Any]) -> dict[str, Any]: + if not isinstance(snapshot, dict): + raise ContractError("delivery snapshot must be an object") + adapter, profile = _validated(snapshot) + binary = Path(adapter["binary"]) + try: + resolved = binary.resolve(strict=True) + except OSError as exc: + raise CapabilityUnavailable("frozen Claude executable is missing") from exc + if (resolved != binary + or hashlib.sha256(binary.read_bytes()).hexdigest() + != adapter["binary_sha256"]): + raise CapabilityUnavailable("frozen Claude executable changed after preflight") + try: + version = subprocess.run( + [str(binary), "--version"], text=True, capture_output=True, + timeout=3, check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise CapabilityUnavailable("Claude version could not be re-observed") from exc + if version.returncode != 0 or version.stdout.strip() != adapter["harness_version"]: + raise CapabilityUnavailable("Claude version changed after preflight") + + workspace = Path(snapshot["delivery_workspace"]["path"]).resolve(strict=True) + effort = profile["effort"]["value"] + model = profile["model_id"] + prompt = build_implementation_prompt( + snapshot["task"], snapshot["delivery_workspace"], + snapshot.get("revision_request"), + ) + argv = [ + str(binary), "--print", "--output-format", "json", "--safe-mode", + "--disable-slash-commands", "--no-session-persistence", + "--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}', + "--no-chrome", "--model", model, "--effort", effort, + *adapter["permission_args"], prompt, + ] + timeout_seconds = snapshot["task"]["budget"]["wall_seconds"] + try: + completed = subprocess.run( + argv, + cwd=workspace, + env={**os.environ, "DEVSQUAD_WORKER": "1"}, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + timeout=timeout_seconds, + check=False, + ) + timed_out = False + except subprocess.TimeoutExpired as exc: + completed = subprocess.CompletedProcess( + argv, 124, exc.stdout or "", exc.stderr or "", + ) + timed_out = True + stdout = completed.stdout.decode("utf-8", "replace") if isinstance( + completed.stdout, bytes + ) else completed.stdout + stderr = completed.stderr.decode("utf-8", "replace") if isinstance( + completed.stderr, bytes + ) else completed.stderr + if (len(stdout.encode()) > MAX_CLAUDE_OUTPUT_BYTES + or len(stderr.encode()) > MAX_CLAUDE_OUTPUT_BYTES): + raise ContractError("Claude implementation output exceeds its byte limit") + import re + error_code = next(( + code for code, pattern in adapter["error_patterns"].items() + if re.search(pattern, stderr, re.IGNORECASE) + ), None) + if timed_out: + error_code = "TIMEOUT" + elif completed.returncode != 0 and error_code is None: + error_code = "CLI_ERROR" + try: + summary, session_id, usage = _structured_result(stdout) + except ContractError as exc: + try: + provider_document = json.loads(stdout) + except json.JSONDecodeError: + provider_document = {} + provider_text = str( + provider_document.get("result") or provider_document.get("error") or "" + ) + error_code = error_code or next(( + code for code, pattern in adapter["error_patterns"].items() + if re.search(pattern, provider_text, re.IGNORECASE) + ), None) + if error_code is None and re.search( + adapter["denied_pattern"], provider_text, re.IGNORECASE, + ): + error_code = "CLI_ERROR" + raise ContractError( + f"{error_code or 'CLI_ERROR'}: Claude implementation failed" + ) from exc + if error_code is not None: + raise ContractError(f"{error_code}: Claude implementation failed") + observed = { + "harness": "claude", + "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], + "model_id": model, + "effort": effort, + "permission_policy": "workspace_write", + "verification": "verified", + } + return make_implementation_evidence( + snapshot, + summary, + observed_identity=observed, + native_ids={"session_id": session_id}, + native_model_requests=None, + usage=usage, + ) + + +def main() -> int: + payload = sys.stdin.buffer.read(MAX_SNAPSHOT_BYTES + 1) + if len(payload) > MAX_SNAPSHOT_BYTES: + raise ContractError("delivery snapshot exceeds its byte limit") + try: + snapshot = json.loads(payload.decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise ContractError("delivery snapshot is not valid UTF-8 JSON") from exc + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/delivery_worker.py b/plugin/core/src/devsquad/delivery_worker.py index 2c87249..76d448c 100644 --- a/plugin/core/src/devsquad/delivery_worker.py +++ b/plugin/core/src/devsquad/delivery_worker.py @@ -38,8 +38,21 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: or index >= len(fixtures)): raise ContractError("offline implementation fixture iteration is missing") fixture = fixtures[index] - if not isinstance(fixture, dict) or set(fixture) != {"writes", "delay_seconds"}: + if (not isinstance(fixture, dict) + or set(fixture) not in ( + {"writes", "delay_seconds"}, + {"writes", "delay_seconds", "fail_profile_ids"}, + )): raise ContractError("offline implementation fixture is incomplete") + fail_profile_ids = fixture.get("fail_profile_ids", []) + if (not isinstance(fail_profile_ids, list) + or not all(isinstance(item, str) and item for item in fail_profile_ids)): + raise ContractError("implementation fixture fail_profile_ids is invalid") + selected_profile_id = snapshot["routing"]["roles"]["implementer"][ + "selected" + ]["profile_id"] + if selected_profile_id in fail_profile_ids: + raise ContractError("RATE_LIMITED: offline implementation fixture failure") writes, delay = fixture["writes"], fixture["delay_seconds"] if (not isinstance(writes, list) or not writes or len(writes) > MAX_FIXTURE_WRITES): diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index b7f2413..48dc5f7 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -245,7 +245,13 @@ def main(argv=None): except ConflictError: pass elif (current["state"] == "queued" and current["phase"] is None - and not (workflow == "issue-delivery" and role == "implementer")): + and not ( + workflow == "issue-delivery" + and role == "implementer" + and isinstance( + json.loads(current["mutable_snapshot"]).get("candidate"), dict, + ) + )): try: Service(Path(args.database).parent).resume(args.run_id) except ConflictError: diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 8fccd60..3147557 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -15,6 +15,7 @@ from .codex_lead_worker import freeze_codex_lead from .codex_review_worker import freeze_codex_reviewer +from .claude_delivery_worker import freeze_claude_implementer from .contracts import ( BudgetExhausted, CapabilityUnavailable, @@ -500,14 +501,26 @@ def _resolve_snapshot( task["scope"]["write_paths"], required_clean_paths=config_paths.values(), ) - if internal_implementation_fixture is None: - raise CapabilityUnavailable( - "live issue-delivery implementer execution is not available yet" + fixture_fields = ( + {"writes", "delay_seconds"}, + {"writes", "delay_seconds", "fail_profile_ids"}, + ) + def valid_implementation_fixture(value: Any) -> bool: + return ( + isinstance(value, dict) + and set(value) in fixture_fields + and ( + "fail_profile_ids" not in value + or isinstance(value["fail_profile_ids"], list) + and all( + isinstance(item, str) and item + for item in value["fail_profile_ids"] + ) + ) ) - valid_fixture = ( - isinstance(internal_implementation_fixture, dict) - and set(internal_implementation_fixture) - == {"writes", "delay_seconds"} + + valid_fixture = valid_implementation_fixture( + internal_implementation_fixture ) if (isinstance(internal_implementation_fixture, dict) and set(internal_implementation_fixture) == {"iterations"}): @@ -517,16 +530,17 @@ def _resolve_snapshot( and 1 <= len(fixtures) <= task["budget"]["max_revisions"] + 1 and all( - isinstance(item, dict) - and set(item) == {"writes", "delay_seconds"} + valid_implementation_fixture(item) for item in fixtures ) ) - if not valid_fixture: + if (internal_implementation_fixture is not None + and not valid_fixture): raise ContractError("internal implementation fixture is invalid") - snapshot["internal_implementation_fixture"] = json.loads( - canonical_json(internal_implementation_fixture) - ) + if internal_implementation_fixture is not None: + snapshot["internal_implementation_fixture"] = json.loads( + canonical_json(internal_implementation_fixture) + ) if internal_review_fixture is not None: if (not isinstance(internal_review_fixture, dict) or set(internal_review_fixture) @@ -606,6 +620,23 @@ def _continue_preparation( internal_implementation_fixture=internal_implementation_fixture, capacity_in_flight=store.active_pool_counts(), ) + if (task["workflow"] == "issue-delivery" + and internal_delay is None + and internal_implementation_fixture is None): + implementer_route = snapshot["routing"]["roles"]["implementer"] + implementer_candidates = [ + implementer_route["selected"], + *implementer_route["fallbacks"], + ] + snapshot["implementation_adapters"] = { + candidate["profile_id"]: freeze_claude_implementer(candidate) + for candidate in implementer_candidates + } + snapshot["implementation_adapter"] = ( + snapshot["implementation_adapters"][ + implementer_route["selected"]["profile_id"] + ] + ) if ((task["workflow"] == "branch-review" or (task["workflow"] == "issue-delivery" and internal_implementation_fixture is None)) diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index d8dc03f..de4a5c1 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -680,11 +680,19 @@ def import_durable(self, run_id: str) -> str: "message": semantic_error, } else: + stderr_text = captures["stderr"].decode("utf-8", "replace") + upper_stderr = stderr_text.upper() + provider_error = next(( + code for code in ("AUTH_ERROR", "RATE_LIMITED") + if f"{code}:" in upper_stderr + ), None) + failure_code = provider_error or ( + "HEADLESS_LEAD_FAILED" + if role == "lead" else "REVIEW_WORKER_FAILED" + if role == "reviewer" else "IMPLEMENTATION_WORKER_FAILED" + ) report_error = { - "error": ( - "HEADLESS_LEAD_FAILED" - if role == "lead" else "REVIEW_WORKER_FAILED" - ), + "error": failure_code, "message": ( "headless lead exited before producing a valid disposition" if role == "lead" diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index cfe640a..3d30efd 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -597,6 +597,11 @@ def validate_implementation_evidence( def make_implementation_evidence( snapshot: dict[str, Any], summary: str, + *, + observed_identity: dict[str, Any] | None = None, + native_ids: dict[str, str] | None = None, + native_model_requests: int | None = None, + usage: dict[str, Any] | None = None, ) -> dict[str, Any]: selected = snapshot["routing"]["roles"]["implementer"]["selected"] document = { @@ -613,11 +618,11 @@ def make_implementation_evidence( snapshot.get("revision_request"), ).encode() ).hexdigest(), - "observed_identity": None, - "native_ids": {}, + "observed_identity": observed_identity, + "native_ids": native_ids if native_ids is not None else {}, "worker_invocations": 1, - "native_model_requests": None, - "usage": { + "native_model_requests": native_model_requests, + "usage": usage if usage is not None else { "input_tokens": None, "output_tokens": None, "total_tokens": None, diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py index a6b2d52..217a235 100644 --- a/test/core/test_delivery_workflow.py +++ b/test/core/test_delivery_workflow.py @@ -2,17 +2,23 @@ import hashlib import json +import os import subprocess import sys import tempfile import time import unittest from pathlib import Path +from unittest.mock import patch CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" sys.path.insert(0, str(CORE / "src")) from devsquad.contracts import ContractError +from devsquad.claude_delivery_worker import ( + freeze_claude_implementer, + run as run_claude_implementer, +) from devsquad.service import Service from devsquad.store import ConflictError, Store, request_hash from devsquad.workspaces import ( @@ -92,6 +98,12 @@ def profile( model="fixture-write-model", permission="workspace_write", ), + profile( + "fixture-implementer-fallback", + family="fixture-family-a2", + model="fixture-write-model-fallback", + permission="workspace_write", + ), profile( "fixture-reviewer", family="fixture-family-b", @@ -113,7 +125,8 @@ def profile( "version": 1, "roles": { "implementer": [ - {"kind": "profile", "id": "fixture-implementer"} + {"kind": "profile", "id": "fixture-implementer"}, + {"kind": "profile", "id": "fixture-implementer-fallback"}, ], "reviewer": [ {"kind": "profile", "id": "fixture-reviewer"} @@ -129,6 +142,11 @@ def profile( "max_concurrency": 1, "unknown_capacity_policy": "allow_bounded", }, + "fixture-implementer-fallback-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, "fixture-reviewer-subscription": { "allowed_billing_modes": ["subscription"], "max_concurrency": 1, @@ -355,6 +373,67 @@ def test_scoped_candidate_commit_patch_and_replay_preserve_source_and_remote(sel self.assertEqual(replay_patch, patch) self.assert_source_unchanged() + def test_frozen_claude_worker_edits_only_the_delivery_workspace(self): + prepared = self.prepare() + binary = self.root / "claude" + binary.write_text( + "#!/bin/sh\n" + "if [ \"$1\" = \"--version\" ]; then\n" + " printf '%s\\n' '2.1.220 (Claude Code)'\n" + " exit 0\n" + "fi\n" + "printf '%s\\n' \"VALUE = 'fixed'\" > src/app.py\n" + "printf '%s\\n' '{\"type\":\"result\",\"is_error\":false," + "\"result\":\"Applied the bounded fix.\"," + "\"session_id\":\"session-fixture\"," + "\"usage\":{\"input_tokens\":12,\"output_tokens\":7}}'\n" + ) + binary.chmod(0o700) + profile = { + "id": "claude-implementer", + "harness": "claude", + "model_family": "claude-sonnet", + "model_id": "claude-sonnet-fixture", + "effort": {"value": "high", "transport": "native"}, + "required_tools": ["read", "write"], + "permission_policy": "workspace_write", + "account_pool_id": "claude-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["fixture"], + } + selected = { + "reference": {"kind": "profile", "id": profile["id"]}, + "binding": None, + "profile_id": profile["id"], + "profile_sha256": hashlib.sha256( + json.dumps(profile, sort_keys=True, separators=(",", ":")).encode() + ).hexdigest(), + "profile": profile, + } + with patch.dict(os.environ, {"PATH": str(self.root)}): + adapter = freeze_claude_implementer(selected) + snapshot = { + "task": self.delivery_task(), + "delivery_workspace": prepared, + "routing": { + "roles": { + "implementer": {"selected": selected, "fallbacks": []}, + }, + }, + "implementation_adapter": adapter, + "implementation_adapters": {profile["id"]: adapter}, + } + evidence = run_claude_implementer(snapshot) + self.assertEqual( + (Path(prepared["path"]) / "src/app.py").read_text(), + "VALUE = 'fixed'\n", + ) + self.assertEqual(evidence["attempt"]["observed_identity"]["harness"], "claude") + self.assertEqual(evidence["attempt"]["native_ids"]["session_id"], "session-fixture") + self.assertEqual(evidence["attempt"]["usage"]["total_tokens"], 19) + self.assert_source_unchanged() + def test_later_mutation_cannot_replay_a_frozen_candidate(self): workspace = Path(self.prepare()["path"]) (workspace / "src/app.py").write_text("VALUE = 'candidate'\n") @@ -773,6 +852,100 @@ def test_headless_delivery_lead_accepts_the_candidate(self): self.assertEqual(len(receipt["lead"]["attempts"]), 1) self.assertEqual(receipt["accounting"]["worker_invocations"], 3) + def test_rate_limited_implementer_uses_frozen_same_permission_fallback(self): + task = self.delivery_task() + task["budget"]["max_fallbacks_per_step"] = 1 + fixture = { + **self.implementation_fixture(), + "fail_profile_ids": ["fixture-implementer"], + } + service = Service(self.runtime) + started = service.start( + task, + "implementation-fallback", + _internal_implementation_fixture=fixture, + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], "fallback-host", + ) + packet = claimed["handoff"]["packet"] + service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision(packet, "accept-fallback", "accept", "Fallback accepted."), + ) + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + implementers = [ + item for item in receipt["attempts"] if item["role"] == "implementer" + ] + self.assertEqual(len(implementers), 2) + self.assertEqual(implementers[0]["status"], "failed") + self.assertEqual(implementers[0]["error"]["error"], "RATE_LIMITED") + self.assertEqual( + [item["selected_profile"]["profile_id"] for item in implementers], + ["fixture-implementer", "fixture-implementer-fallback"], + ) + self.assertEqual( + {item["selected_profile"]["profile"]["permission_policy"] + for item in implementers}, + {"workspace_write"}, + ) + + def test_cancelled_repair_retains_prior_candidate_attempts_and_disposition(self): + fixtures = self.repair_fixtures() + fixtures["iterations"][1]["delay_seconds"] = 5 + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "cancelled-repair", + _internal_implementation_fixture=fixtures, + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], "cancel-repair-host", + ) + packet = claimed["handoff"]["packet"] + service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision(packet, "cancel-repair", "revise", "Repair then cancel."), + ) + deadline = time.monotonic() + 10 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "running": + break + time.sleep(0.02) + else: + self.fail("repair implementer never entered running state") + service.cancel(started["run_id"]) + terminal = self.wait_for_terminal(service, started["run_id"]) + self.assertEqual(terminal["state"], "cancelled") + receipt_artifact = next( + item for item in service.result(started["run_id"])["artifacts"] + if item["name"] == "receipt.json" + ) + receipt = json.loads(Path(receipt_artifact["path"]).read_text()) + self.assertEqual(receipt["workflow"], "issue-delivery") + self.assertEqual( + [item["disposition"] for item in receipt["dispositions"]], ["revise"], + ) + self.assertEqual( + [item["role"] for item in receipt["attempts"]], + ["implementer", "reviewer", "implementer"], + ) + self.assertEqual(receipt["attempts"][-1]["status"], "cancelled") + self.assertEqual(receipt["candidate"]["sha256"], packet["candidate_sha256"]) + def test_live_implementer_cannot_be_resumed_into_a_second_writer(self): service = Service(self.runtime) started = service.start( From f199cd2a45db48a5603de4b48da165356d4257dc Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 27 Sep 2026 08:06:29 -0700 Subject: [PATCH 093/197] test: prove revised writer crash fencing --- test/core/test_delivery_workflow.py | 59 +++++++++++++++++++++++++++++ 1 file changed, 59 insertions(+) diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py index 217a235..3a1aa8f 100644 --- a/test/core/test_delivery_workflow.py +++ b/test/core/test_delivery_workflow.py @@ -3,6 +3,7 @@ import hashlib import json import os +import signal import subprocess import sys import tempfile @@ -946,6 +947,64 @@ def test_cancelled_repair_retains_prior_candidate_attempts_and_disposition(self) self.assertEqual(receipt["attempts"][-1]["status"], "cancelled") self.assertEqual(receipt["candidate"]["sha256"], packet["candidate_sha256"]) + def test_killed_repair_supervisor_never_launches_a_duplicate_writer(self): + fixtures = self.repair_fixtures() + fixtures["iterations"][1]["delay_seconds"] = 3 + service = Service(self.runtime) + started = service.start( + self.delivery_task(), + "killed-repair-supervisor", + _internal_implementation_fixture=fixtures, + _internal_review_fixture=self.clean_review_fixture(), + ) + self.wait_for_candidate(service, started["run_id"]) + service.resume(started["run_id"]) + waiting = self.wait_for_handoff(service, started["run_id"]) + claimed = service.handoff_claim( + started["run_id"], waiting["version"], "kill-repair-host", + ) + packet = claimed["handoff"]["packet"] + service.handoff_complete( + started["run_id"], claimed["claim"], + self.decision(packet, "kill-repair", "revise", "Repair candidate."), + ) + deadline = time.monotonic() + 10 + attempt = None + while time.monotonic() < deadline: + store = Store(service.database, service.artifacts) + try: + current = store.attempt(started["run_id"]) + if (current and current["status"] == "running" + and current["role"] == "implementer" + and Path(current["child_record"]).is_file()): + attempt = current + break + finally: + store.close() + time.sleep(0.02) + if attempt is None: + self.fail("repair writer did not publish its child identity") + os.kill(attempt["pid"], signal.SIGKILL) + time.sleep(0.1) + recovered = Service(self.runtime).resume( + started["run_id"], + {"attempt_id": attempt["id"], "disposition": "retain_ownership"}, + ) + self.assertFalse(recovered["launched"]) + self.assertEqual(recovered["disposition"], "retain_ownership") + store = Store(service.database, service.artifacts) + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual( + [item["role"] for item in attempts], + ["implementer", "reviewer", "implementer"], + ) + finally: + store.close() + cancelled = service.cancel(started["run_id"]) + self.assertEqual(cancelled["state"], "cancelled") + self.assert_source_unchanged() + def test_live_implementer_cannot_be_resumed_into_a_second_writer(self): service = Service(self.runtime) started = service.start( From 41c5f1bce7cc9fda78a01bce833373413c166725 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 27 Sep 2026 08:12:36 -0700 Subject: [PATCH 094/197] docs: record M5 offline completion --- docs/plans/engineering-team/M5-STATUS.md | 45 ++++++++++++++++++------ docs/plans/engineering-team/RESUME.md | 37 +++++++++++-------- docs/plans/engineering-team/backlog.json | 13 +++++-- 3 files changed, 69 insertions(+), 26 deletions(-) diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index a15ef79..4fbdf40 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -1,22 +1,22 @@ # M5 implementation status -M5 is **in progress**. This matrix is derived from the authoritative M5 -requirements before implementation; a row becomes verified only when its -behavioral evidence exists. The milestone remains incomplete until the live -two-harness gate passes. +M5 is **blocked on its external live gate**. All independently executable +offline implementation and fault-injection work is complete; the milestone +remains incomplete until a normally authenticated Claude implementation and a +different-model Codex review pass the live two-harness gate. | Requirement | Planned evidence | Status | |---|---|---| | Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | verified offline at `d96e9e4` | | Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | verified offline at `0e88d73` | | Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `0e88d73` | -| Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | router verified; workflow pending | +| Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | workflow and identity gate verified offline at `b7d90cc`; live proof pending | | Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | verified offline through two distinct candidates at `4c76887` | -| Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | correction and revision/invocation budgets verified offline at `4c76887`; delivery fallback faults pending | -| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | mandatory-check gate and accept/reject receipts verified offline at `a1199c6`; invalid-output fault matrix pending | -| Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | successful repair and headless history verified offline at `4c76887`; failed fallback/cancellation history pending | -| Crash recovery | Killing a live implementation supervisor cannot create a duplicate writer on resume | initial live-writer and revised prelaunch recovery verified offline at `4c76887`; live revised-writer fault gate pending | -| Live acceptance | One bounded issue completes across at least two authenticated subscription harnesses with different verified models | pending | +| Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | correction/budgets verified at `4c76887`; same-permission delivery fallback verified at `b7d90cc` | +| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | verified offline at `a1199c6` and `b7d90cc` | +| Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | success, repair, fallback, failure and cancellation history verified offline at `b7d90cc` | +| Crash recovery | Killing a live implementation supervisor cannot create a duplicate writer on resume | prelaunch and live revised-writer recovery verified offline at `f199cd2` | +| Live acceptance | One bounded issue completes across at least two authenticated subscription harnesses with different verified models | blocked on normal Claude CLI login | ## Boundary @@ -115,3 +115,28 @@ terminalizes through its own fenced attempt. The complete gate is **237 core tests (2 optional-SDK skips)** with `ResourceWarning` promoted to error and **220 Bash assertions**. Delivery fallback/failure/cancellation history, live Claude implementation and the live two-harness proof remain open. + +## Plan 07-02 checkpoint 3 + +At `b7d90cc`, the frozen delivery package can invoke the exact preflighted +Claude binary without importing source-tree adapter resources at runtime. It +retains the requested and observed identity, native session and usage evidence, +classifies provider faults through the frozen contract, and permits only the +predeclared same-permission fallback. A rate-limited implementer therefore +uses the one frozen fallback without widening tools or permissions, and the +terminal receipt retains both attempts. Delivery-specific cancellation keeps +the prior candidate, attempts and lead disposition visible instead of +collapsing history. + +At `f199cd2`, a real controlled subprocess kills the delivery supervisor while +a revised implementation child is live. Recovery reports retained ownership, +does not launch a duplicate writer, and cancellation reaps the process while +the source checkout and remote refs remain unchanged. The exact checkpoint +passes **241 core tests with 2 optional-SDK skips** and ResourceWarning promoted +to error. The compatibility gate remains **220/220 Bash assertions**; the last +code change after that run added only the delivery-specific Python regression. + +No independent M5 implementation work remains. The unresolved acceptance gate +is intentionally not replaced with fixture evidence: normal Claude CLI login +is required to run a genuine bounded implementation followed by different- +model Codex review/check/disposition and save the redacted two-harness receipt. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 5547ff7..53fa4e6 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,7 +2,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 26, 2026 +## Current position — September 27, 2026 - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` remains the published runtime @@ -121,6 +121,16 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne tests (2 optional-SDK skips) and 220 Bash assertions. Delivery fallback, failure/cancellation history, live Claude execution and the live two-harness proof remain open. +- M5 has completed all independently executable offline work at `f199cd2`. + `b7d90cc` freezes the real Claude implementation bridge, records observed + identity/session/usage, enforces the same-permission rate-limit fallback and + preserves delivery failure/cancellation history. `f199cd2` kills a live + revised-implementation supervisor and proves retained ownership, no duplicate + writer, successful reap and unchanged source checkout/remotes. The exact core + gate is 241 tests with 2 optional-SDK skips and ResourceWarning promoted to + error; the compatibility gate is 220 Bash assertions. M5 is now blocked only + on normal Claude login for the genuine Claude implementation → different- + model Codex review/check/disposition receipt. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -147,7 +157,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 231 discovered through the prepared Jev pilot; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | +| Python core discovery | 241 discovered through M5 live-writer recovery; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -175,8 +185,8 @@ the earlier apparent nonresponses. The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md), [M2-STATUS.md](M2-STATUS.md) and [M3-STATUS.md](M3-STATUS.md). -[backlog.json](backlog.json) marks M1–M3 complete, M4 blocked on its external -Claude live gate, and M5 in progress. M6 classifier work packages remain pending. +[backlog.json](backlog.json) marks M1–M3 complete and M4/M5 blocked only on +their external Claude live gates. M6 implementation work remains pending. Unauthenticated, unsupported or permission-blocked provider paths are not advertised as verified. @@ -184,19 +194,18 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M5 Plan 07-02 Task 3 from `4c76887`: complete delivery-specific - implementer/reviewer fallback faults, deadline/cancellation terminal history - and remaining live-revised-writer recovery. Then wire the already-conformed - Claude adapter into live implementation and run the live two-harness gate - only after normal provider login. Accept/reject, seeded repair, stale - evidence, required-check enforcement, finite revision/invocation budgets, - headless lead and prelaunch crash recovery are green; do not rebuild them. +2. Begin M6 with the shared-capacity observation/reservation model: persist all + applicable windows, derive available/exhausted/unknown from fresh evidence, + fence concurrent reservations, expose the snapshot through the service/CLI + and feed it into deterministic routing. Then add evidence-based outcomes, + lifecycle qualification/guarded promotion and the default-off decision + helper. Do not rebuild the completed M5 offline path. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. -4. When M5's evidence shape is stable, execute the small M6 decision-helper - work packages alongside other independently ready M6 work. Keep experiments - off by default and preserve all existing M4/M5/live acceptance gates. +4. Execute the small M6 decision-helper work packages alongside other + independently ready M6 work. Keep experiments off by default and preserve + all existing M4/M5/live acceptance gates. Exception already authorized: once TypeSafe login/API-key setup is complete, run the prepared one-request synthetic Jev pilot immediately, save the private receipt outside Git, and update only redacted aggregate evidence. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 6475663..ca97de3 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -211,7 +211,7 @@ "id": "M5", "title": "Bounded implementation with independent review", "depends_on": ["M3"], - "status": "in_progress", + "status": "blocked", "acceptance_section": "M5 — Deliver a bounded engineering change", "evidence": [ { @@ -258,9 +258,18 @@ "artifact": "M5-STATUS.md", "recorded_at": "2026-09-26T23:58:00-07:00", "availability": "tracked_tests" + }, + { + "kind": "offline_completion_checkpoint", + "revision": "f199cd2", + "command_or_action": "241 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, 19 focused delivery regressions and a controlled live-child supervisor kill", + "outcome": "All independent M5 behavior is implemented offline: the frozen Claude bridge records identity/session/usage, same-permission rate-limit fallback and cancellation retain complete history, and revised live-writer recovery cannot launch a duplicate or mutate the source checkout/remotes. Only the genuine authenticated Claude-to-Codex two-harness acceptance receipt remains.", + "artifact": "M5-STATUS.md", + "recorded_at": "2026-09-27T08:10:15-07:00", + "availability": "tracked_tests" } ], - "blocker": null + "blocker": "Normal Claude CLI login is required for the genuine bounded Claude implementation followed by different-model Codex review/check/disposition; offline fixtures are not substituted for the two-harness acceptance receipt" }, { "id": "M6", From 21b133141f39b36a470efe559a6c99a3b9d066a9 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 27 Sep 2026 08:23:28 -0700 Subject: [PATCH 095/197] feat: define shared capacity evidence --- plugin/core/src/devsquad/capacity.py | 262 ++++++++++++++++++ .../src/devsquad/migrations/009_capacity.sql | 42 +++ plugin/core/src/devsquad/store.py | 2 +- test/core/test_capacity.py | 179 ++++++++++++ test/core/test_cli.py | 4 +- test/core/test_handoff_store.py | 6 +- test/core/test_store.py | 8 +- 7 files changed, 493 insertions(+), 10 deletions(-) create mode 100644 plugin/core/src/devsquad/capacity.py create mode 100644 plugin/core/src/devsquad/migrations/009_capacity.sql create mode 100644 test/core/test_capacity.py diff --git a/plugin/core/src/devsquad/capacity.py b/plugin/core/src/devsquad/capacity.py new file mode 100644 index 0000000..633461a --- /dev/null +++ b/plugin/core/src/devsquad/capacity.py @@ -0,0 +1,262 @@ +"""Strict shared-capacity observations and deterministic availability derivation.""" + +from __future__ import annotations + +from datetime import datetime, timedelta, timezone +import json +import math +from typing import Any + +from .contracts import ContractError +from .store import canonical_json + + +OBSERVATION_FIELDS = { + "schema_version", + "observation_id", + "pool_id", + "window_id", + "applies_to", + "observed_at", + "expires_at", + "source", + "used", + "limit", + "unit", + "resets_at", + "confidence", +} +APPLIES_TO_FIELDS = {"harnesses", "model_families", "model_ids"} +SOURCES = {"native_reported", "manual_reported", "estimated"} +UNITS = {"percent", "requests", "tokens", "provider-native-string"} +CONFIDENCE = {"confirmed", "reported", "estimated"} +CAPACITY_STATES = {"available", "exhausted", "unknown"} +MAX_CLOCK_SKEW = timedelta(minutes=5) + + +def _authoritative_now(value: datetime | None) -> datetime: + current = datetime.now(timezone.utc) if value is None else value + if not isinstance(current, datetime) or current.tzinfo is None or current.utcoffset() is None: + raise ContractError("capacity evaluation time must include a timezone") + return current.astimezone(timezone.utc) + + +def _timestamp(value: Any, field: str, *, nullable: bool = False) -> datetime | None: + if value is None and nullable: + return None + if not isinstance(value, str) or not value: + suffix = " or null" if nullable else "" + raise ContractError(f"capacity {field} must be an ISO timestamp{suffix}") + try: + parsed = datetime.fromisoformat(value) + except ValueError as exc: + raise ContractError(f"capacity {field} must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError(f"capacity {field} must include a timezone") + return parsed.astimezone(timezone.utc) + + +def _identifier(value: Any, field: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ContractError(f"capacity {field} must be a non-empty string") + return value + + +def _measurement(value: Any, field: str) -> int | float | None: + if value is None: + return None + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ContractError(f"capacity {field} must be a finite non-negative number or null") + if not math.isfinite(value) or value < 0: + raise ContractError(f"capacity {field} must be a finite non-negative number or null") + return value + + +def _selector(value: Any) -> dict[str, list[str]]: + if not isinstance(value, dict) or set(value) != APPLIES_TO_FIELDS: + raise ContractError("capacity applies_to fields are invalid") + normalized: dict[str, list[str]] = {} + for field in sorted(APPLIES_TO_FIELDS): + entries = value[field] + if not isinstance(entries, list): + raise ContractError(f"capacity applies_to.{field} must be an array") + if any(not isinstance(item, str) or not item for item in entries): + raise ContractError( + f"capacity applies_to.{field} must contain non-empty strings", + ) + if len(set(entries)) != len(entries): + raise ContractError(f"capacity applies_to.{field} must be unique") + normalized[field] = sorted(entries) + return normalized + + +def validate_observation( + value: dict[str, Any], *, now: datetime | None = None, +) -> dict[str, Any]: + """Validate and normalize one schema-v1 capacity observation.""" + if not isinstance(value, dict) or set(value) != OBSERVATION_FIELDS: + unknown = sorted(set(value) - OBSERVATION_FIELDS) if isinstance(value, dict) else [] + missing = sorted(OBSERVATION_FIELDS - set(value)) if isinstance(value, dict) else [] + raise ContractError( + f"capacity observation fields invalid: unknown={unknown} missing={missing}", + ) + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("capacity observation schema_version is invalid") + normalized = { + **value, + "observation_id": _identifier(value["observation_id"], "observation_id"), + "pool_id": _identifier(value["pool_id"], "pool_id"), + "window_id": _identifier(value["window_id"], "window_id"), + "applies_to": _selector(value["applies_to"]), + } + if not isinstance(value["source"], str) or value["source"] not in SOURCES: + raise ContractError("capacity source is invalid") + if not isinstance(value["unit"], str) or value["unit"] not in UNITS: + raise ContractError("capacity unit is invalid") + if not isinstance(value["confidence"], str) or value["confidence"] not in CONFIDENCE: + raise ContractError("capacity confidence is invalid") + normalized["used"] = _measurement(value["used"], "used") + normalized["limit"] = _measurement(value["limit"], "limit") + if value["unit"] == "percent" and any( + item is not None and item > 100 + for item in (normalized["used"], normalized["limit"]) + ): + raise ContractError("capacity percent measurements must be at most 100") + observed = _timestamp(value["observed_at"], "observed_at") + expires = _timestamp(value["expires_at"], "expires_at") + _timestamp(value["resets_at"], "resets_at", nullable=True) + if observed > expires: + raise ContractError("capacity observed_at must not be after expires_at") + current = _authoritative_now(now) + if observed > current + MAX_CLOCK_SKEW: + raise ContractError("capacity observed_at exceeds allowed clock skew") + # A canonical round trip proves every retained value is finite JSON and + # prevents callers from mutating the source object after validation. + return json.loads(canonical_json(normalized)) + + +def _matches(selector: dict[str, list[str]], target: dict[str, Any] | None) -> bool: + mapping = { + "harnesses": "harness", + "model_families": "model_family", + "model_ids": "model_id", + } + for selector_field, target_field in mapping.items(): + allowed = selector[selector_field] + if not allowed: + continue + if target is None or target.get(target_field) not in allowed: + return False + return True + + +def derive_pool_capacity( + pool_id: str, + observations: list[dict[str, Any]], + *, + target: dict[str, Any] | None = None, + in_flight: int = 0, + now: datetime | None = None, +) -> dict[str, Any]: + """Derive hard availability from the latest applicable observation per window.""" + _identifier(pool_id, "pool_id") + if not isinstance(observations, list): + raise ContractError("capacity observations must be an array") + if type(in_flight) is not int or in_flight < 0: + raise ContractError("capacity in_flight must be a non-negative integer") + if target is not None: + if not isinstance(target, dict) or any( + not isinstance(target.get(field), str) or not target[field] + for field in ("harness", "model_family", "model_id") + ): + raise ContractError("capacity target identity is invalid") + current = _authoritative_now(now) + latest: dict[tuple[str, str], tuple[datetime, str, dict[str, Any]]] = {} + seen_ids: set[str] = set() + for raw in observations: + observation = validate_observation(raw, now=current) + observation_id = observation["observation_id"] + if observation_id in seen_ids: + raise ContractError("capacity observation ids must be unique") + seen_ids.add(observation_id) + if observation["pool_id"] != pool_id or not _matches( + observation["applies_to"], target, + ): + continue + key = (observation["window_id"], canonical_json(observation["applies_to"])) + ordering = ( + _timestamp(observation["observed_at"], "observed_at"), + observation_id, + ) + prior = latest.get(key) + if prior is None or ordering[:2] > prior[:2]: + latest[key] = (ordering[0], ordering[1], observation) + + windows = [] + for key in sorted(latest): + observation = latest[key][2] + fresh = current <= _timestamp(observation["expires_at"], "expires_at") + authoritative = ( + observation["source"] != "estimated" + and observation["confidence"] != "estimated" + ) + if not fresh: + status, reason = "unknown", "stale_observation" + elif not authoritative: + status, reason = "unknown", "estimated_observation" + elif observation["used"] is None or observation["limit"] is None: + status, reason = "unknown", "unknown_measurement" + elif observation["used"] >= observation["limit"]: + status, reason = "exhausted", "window_exhausted" + else: + status, reason = "available", "window_available" + windows.append({ + "observation_id": observation["observation_id"], + "window_id": observation["window_id"], + "applies_to": observation["applies_to"], + "observed_at": observation["observed_at"], + "expires_at": observation["expires_at"], + "resets_at": observation["resets_at"], + "source": observation["source"], + "confidence": observation["confidence"], + "used": observation["used"], + "limit": observation["limit"], + "unit": observation["unit"], + "fresh": fresh, + "authoritative": authoritative, + "status": status, + "reason": reason, + }) + + if any(window["status"] == "exhausted" for window in windows): + status = "exhausted" + elif not windows or any(window["status"] == "unknown" for window in windows): + status = "unknown" + else: + status = "available" + observed_at = None + if windows: + observed_at = max( + windows, + key=lambda window: _timestamp(window["observed_at"], "observed_at"), + )["observed_at"] + reasons = [ + f"{window['reason']}:{window['window_id']}" + for window in windows + if window["status"] != "available" + ] + if not windows: + reasons = ["no_applicable_observations"] + return { + "schema_version": 1, + "pool_id": pool_id, + "status": status, + "in_flight": in_flight, + "observed_at": observed_at, + "evaluated_at": current.isoformat(), + "target": None if target is None else { + field: target[field] for field in ("harness", "model_family", "model_id") + }, + "windows": windows, + "reasons": reasons, + } diff --git a/plugin/core/src/devsquad/migrations/009_capacity.sql b/plugin/core/src/devsquad/migrations/009_capacity.sql new file mode 100644 index 0000000..3ee18fb --- /dev/null +++ b/plugin/core/src/devsquad/migrations/009_capacity.sql @@ -0,0 +1,42 @@ +CREATE TABLE pool_observations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + observation_id TEXT NOT NULL UNIQUE, + pool_id TEXT NOT NULL, + window_id TEXT NOT NULL, + applies_to_json TEXT NOT NULL, + observed_at TEXT NOT NULL, + expires_at TEXT NOT NULL, + source TEXT NOT NULL CHECK(source IN ('native_reported', 'manual_reported', 'estimated')), + used REAL, + limit_value REAL, + unit TEXT NOT NULL CHECK(unit IN ('percent', 'requests', 'tokens', 'provider-native-string')), + resets_at TEXT, + confidence TEXT NOT NULL CHECK(confidence IN ('confirmed', 'reported', 'estimated')), + recorded_at TEXT NOT NULL, + CHECK(used IS NULL OR used >= 0), + CHECK(limit_value IS NULL OR limit_value >= 0) +); + +CREATE INDEX pool_observations_lookup +ON pool_observations(pool_id, window_id, observed_at); + +CREATE TABLE pool_reservations ( + id TEXT PRIMARY KEY, + pool_id TEXT NOT NULL, + run_id TEXT NOT NULL REFERENCES runs(id), + attempt_id TEXT REFERENCES attempts(id), + purpose TEXT NOT NULL CHECK(purpose IN ('attempt', 'qualification', 'classifier')), + profile_id TEXT, + reserved_at TEXT NOT NULL, + reconciled_at TEXT, + reconcile_reason TEXT, + UNIQUE(attempt_id), + CHECK( + (reconciled_at IS NULL AND reconcile_reason IS NULL) + OR (reconciled_at IS NOT NULL AND reconcile_reason IS NOT NULL) + ) +); + +CREATE INDEX pool_reservations_active +ON pool_reservations(pool_id, reserved_at) +WHERE reconciled_at IS NULL; diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 1cb6d48..a01d66f 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 8 +SUPPORTED_SCHEMA_VERSION = 9 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py new file mode 100644 index 0000000..31634b8 --- /dev/null +++ b/test/core/test_capacity.py @@ -0,0 +1,179 @@ +import copy +from datetime import datetime, timedelta, timezone +import math +from pathlib import Path +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.capacity import derive_pool_capacity, validate_observation +from devsquad.contracts import ContractError +from devsquad.store import Store + + +NOW = datetime(2026, 9, 27, 15, 0, tzinfo=timezone.utc) + + +def observation( + observation_id="obs-short", + *, + window_id="short", + observed_at=NOW - timedelta(minutes=1), + expires_at=NOW + timedelta(minutes=10), + source="native_reported", + used=20, + limit=100, + unit="percent", + confidence="confirmed", + applies_to=None, +): + return { + "schema_version": 1, + "observation_id": observation_id, + "pool_id": "shared-pool", + "window_id": window_id, + "applies_to": applies_to or { + "harnesses": [], + "model_families": [], + "model_ids": [], + }, + "observed_at": observed_at.isoformat(), + "expires_at": expires_at.isoformat(), + "source": source, + "used": used, + "limit": limit, + "unit": unit, + "resets_at": (NOW + timedelta(hours=1)).isoformat(), + "confidence": confidence, + } + + +class CapacityContractTest(unittest.TestCase): + def test_validation_is_strict_and_normalizes_scopes(self): + value = observation(applies_to={ + "harnesses": ["grok", "codex"], + "model_families": [], + "model_ids": ["model-b", "model-a"], + }) + normalized = validate_observation(value, now=NOW) + self.assertEqual(normalized["applies_to"]["harnesses"], ["codex", "grok"]) + self.assertEqual(normalized["applies_to"]["model_ids"], ["model-a", "model-b"]) + value["applies_to"]["harnesses"].append("claude") + self.assertEqual(normalized["applies_to"]["harnesses"], ["codex", "grok"]) + + invalid = copy.deepcopy(normalized) + invalid["extra"] = True + with self.assertRaisesRegex(ContractError, "fields invalid"): + validate_observation(invalid, now=NOW) + invalid = observation(used=True) + with self.assertRaisesRegex(ContractError, "finite non-negative"): + validate_observation(invalid, now=NOW) + invalid = observation(used=math.inf) + with self.assertRaisesRegex(ContractError, "finite non-negative"): + validate_observation(invalid, now=NOW) + invalid = observation(used=101) + with self.assertRaisesRegex(ContractError, "at most 100"): + validate_observation(invalid, now=NOW) + invalid = observation() + invalid["source"] = [] + with self.assertRaisesRegex(ContractError, "source"): + validate_observation(invalid, now=NOW) + invalid = observation(observed_at=NOW + timedelta(minutes=6)) + invalid["expires_at"] = (NOW + timedelta(minutes=20)).isoformat() + with self.assertRaisesRegex(ContractError, "clock skew"): + validate_observation(invalid, now=NOW) + invalid = observation() + invalid["observed_at"] = "2026-09-27T15:00:00" + with self.assertRaisesRegex(ContractError, "timezone"): + validate_observation(invalid, now=NOW) + + def test_fresh_exhausted_weekly_window_beats_available_short_window(self): + short = observation() + weekly = observation( + "obs-weekly", window_id="weekly", used=100, limit=100, + ) + snapshot = derive_pool_capacity( + "shared-pool", [short, weekly], in_flight=1, now=NOW, + ) + self.assertEqual(snapshot["status"], "exhausted") + self.assertEqual(snapshot["in_flight"], 1) + self.assertEqual( + [window["status"] for window in snapshot["windows"]], + ["available", "exhausted"], + ) + self.assertEqual(snapshot["reasons"], ["window_exhausted:weekly"]) + + def test_stale_estimated_or_unknown_window_never_becomes_zero(self): + stale = observation( + "obs-weekly", window_id="weekly", + expires_at=NOW - timedelta(seconds=1), used=100, limit=100, + ) + snapshot = derive_pool_capacity( + "shared-pool", [observation(), stale], now=NOW, + ) + self.assertEqual(snapshot["status"], "unknown") + self.assertIn("stale_observation:weekly", snapshot["reasons"]) + + estimated = observation( + "obs-estimated", source="estimated", confidence="estimated", + used=100, limit=100, + ) + snapshot = derive_pool_capacity("shared-pool", [estimated], now=NOW) + self.assertEqual(snapshot["status"], "unknown") + self.assertFalse(snapshot["windows"][0]["authoritative"]) + + unknown = observation("obs-unknown", used=None, limit=None) + snapshot = derive_pool_capacity("shared-pool", [unknown], now=NOW) + self.assertEqual(snapshot["status"], "unknown") + self.assertEqual(snapshot["windows"][0]["reason"], "unknown_measurement") + + def test_latest_observation_per_scoped_window_is_used(self): + scope = { + "harnesses": ["codex"], + "model_families": ["gpt"], + "model_ids": [], + } + older = observation( + "older", used=100, limit=100, applies_to=scope, + observed_at=NOW - timedelta(minutes=2), + ) + newer = observation( + "newer", used=10, limit=100, applies_to=scope, + observed_at=NOW - timedelta(minutes=1), + ) + target = {"harness": "codex", "model_family": "gpt", "model_id": "gpt-5"} + snapshot = derive_pool_capacity( + "shared-pool", [older, newer], target=target, now=NOW, + ) + self.assertEqual(snapshot["status"], "available") + self.assertEqual([item["observation_id"] for item in snapshot["windows"]], ["newer"]) + + other = {"harness": "claude", "model_family": "claude", "model_id": "opus"} + snapshot = derive_pool_capacity( + "shared-pool", [older, newer], target=other, now=NOW, + ) + self.assertEqual(snapshot["status"], "unknown") + self.assertEqual(snapshot["reasons"], ["no_applicable_observations"]) + + def test_migration_nine_creates_capacity_ledger(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + version = store.connection.execute( + "SELECT MAX(version) FROM schema_migrations", + ).fetchone()[0] + self.assertEqual(version, 9) + tables = { + row[0] for row in store.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table'", + ) + } + self.assertTrue({"pool_observations", "pool_reservations"} <= tables) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_cli.py b/test/core/test_cli.py index fb8fa77..45b96ee 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -356,7 +356,7 @@ def build_python(): return candidate return None - def test_installed_wheel_contains_and_applies_migrations_through_eight(self): + def test_installed_wheel_contains_and_applies_migrations_through_nine(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") @@ -398,7 +398,7 @@ def test_installed_wheel_contains_and_applies_migrations_through_eight(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 8 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 9 attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index b15e908..286795f 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 8, + 9, ) tables = { row[0] @@ -618,7 +618,7 @@ def build_python(): return candidate return None - def test_installed_wheel_applies_schema_four_to_eight(self): + def test_installed_wheel_applies_schema_four_to_nine(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel") @@ -684,7 +684,7 @@ def test_installed_wheel_applies_schema_four_to_eight(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 8 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 9 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_store.py b/test/core/test_store.py index 98dd45c..298685a 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -412,8 +412,8 @@ def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): ) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 8) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(9,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 9) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(10,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -428,7 +428,7 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 8) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 9) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") @@ -445,7 +445,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],8) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],9) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From 26ce5cf1e8affc76f82a5003b8cc92caa7eb53bf Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 27 Sep 2026 08:31:09 -0700 Subject: [PATCH 096/197] feat: persist and fence shared capacity --- .../src/devsquad/migrations/009_capacity.sql | 25 ++ plugin/core/src/devsquad/store.py | 319 +++++++++++++++++- test/core/test_capacity.py | 154 ++++++++- test/core/test_store.py | 70 ++++ 4 files changed, 549 insertions(+), 19 deletions(-) diff --git a/plugin/core/src/devsquad/migrations/009_capacity.sql b/plugin/core/src/devsquad/migrations/009_capacity.sql index 3ee18fb..f5055db 100644 --- a/plugin/core/src/devsquad/migrations/009_capacity.sql +++ b/plugin/core/src/devsquad/migrations/009_capacity.sql @@ -40,3 +40,28 @@ CREATE TABLE pool_reservations ( CREATE INDEX pool_reservations_active ON pool_reservations(pool_id, reserved_at) WHERE reconciled_at IS NULL; + +INSERT INTO pool_reservations( + id, pool_id, run_id, attempt_id, purpose, profile_id, reserved_at +) +SELECT + 'migrated-attempt-' || id, + account_pool_id, + run_id, + id, + 'attempt', + profile_id, + created_at +FROM attempts +WHERE account_pool_id IS NOT NULL + AND status IN ('reserved', 'running', 'cancelling', 'ownership_ambiguous'); + +CREATE TRIGGER reconcile_pool_reservation_after_attempt_status +AFTER UPDATE OF status ON attempts +WHEN NEW.status NOT IN ('reserved', 'running', 'cancelling', 'ownership_ambiguous') +BEGIN + UPDATE pool_reservations + SET reconciled_at = COALESCE(NEW.finished_at, NEW.heartbeat_at), + reconcile_reason = 'attempt_status_' || NEW.status + WHERE attempt_id = NEW.id AND reconciled_at IS NULL; +END; diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index a01d66f..a2fe1b3 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -180,9 +180,16 @@ def migrate(self) -> None: if len(candidates) != 1: raise SchemaVersionError(f"migration {next_version} is missing or ambiguous") sql = candidates[0].read_text() - for statement in sql.split(";"): - if statement.strip(): - self.connection.execute(statement) + statement = "" + for line in sql.splitlines(keepends=True): + statement += line + if sqlite3.complete_statement(statement): + self.connection.execute(statement.strip()) + statement = "" + if statement.strip(): + raise SchemaVersionError( + f"migration {next_version} contains incomplete SQL", + ) self.connection.execute("INSERT INTO schema_migrations(version, applied_at) VALUES(?, ?)", (next_version, _utc_now())) current = next_version self.connection.execute("COMMIT") @@ -691,14 +698,286 @@ def _enforce_attempt_budget( if wall_seconds * 1000 - elapsed_ms < 1000: raise BudgetExhausted("run wall-time budget is exhausted") + @staticmethod + def _pool_observation(row: sqlite3.Row) -> dict[str, Any]: + return { + "schema_version": 1, + "observation_id": row["observation_id"], + "pool_id": row["pool_id"], + "window_id": row["window_id"], + "applies_to": json.loads(row["applies_to_json"]), + "observed_at": row["observed_at"], + "expires_at": row["expires_at"], + "source": row["source"], + "used": row["used"], + "limit": row["limit_value"], + "unit": row["unit"], + "resets_at": row["resets_at"], + "confidence": row["confidence"], + } + + def _pool_observations(self, pool_id: str) -> list[dict[str, Any]]: + rows = self.connection.execute( + "SELECT observation_id,pool_id,window_id,applies_to_json,observed_at," + "expires_at,source,used,limit_value,unit,resets_at,confidence " + "FROM pool_observations WHERE pool_id=? " + "ORDER BY observed_at,observation_id", + (pool_id,), + ).fetchall() + return [self._pool_observation(row) for row in rows] + + def record_pool_observation( + self, observation: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """Persist one immutable, replay-safe capacity observation.""" + from .capacity import validate_observation + + current = _authoritative_now(now) + normalized = validate_observation(observation, now=current) + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT observation_id,pool_id,window_id,applies_to_json,observed_at," + "expires_at,source,used,limit_value,unit,resets_at,confidence,recorded_at " + "FROM pool_observations WHERE observation_id=?", + (normalized["observation_id"],), + ).fetchone() + if existing is not None: + stored = self._pool_observation(existing) + if stored != normalized: + raise ConflictError( + "capacity observation id was already used with different evidence", + ) + self.connection.execute("COMMIT") + return { + "observation": stored, + "recorded_at": existing["recorded_at"], + "replayed": True, + } + recorded_at = current.isoformat() + self.connection.execute( + "INSERT INTO pool_observations(" + "observation_id,pool_id,window_id,applies_to_json,observed_at,expires_at," + "source,used,limit_value,unit,resets_at,confidence,recorded_at" + ") VALUES(?,?,?,?,?,?,?,?,?,?,?,?,?)", + ( + normalized["observation_id"], normalized["pool_id"], + normalized["window_id"], canonical_json(normalized["applies_to"]), + normalized["observed_at"], normalized["expires_at"], + normalized["source"], normalized["used"], normalized["limit"], + normalized["unit"], normalized["resets_at"], + normalized["confidence"], recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "observation": normalized, + "recorded_at": recorded_at, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def capacity_snapshot( + self, + pool_id: str, + *, + target: dict[str, Any] | None = None, + now: datetime | None = None, + ) -> dict[str, Any]: + """Return current evidence and local reservations for one pool/target.""" + from .capacity import derive_pool_capacity + + current = _authoritative_now(now) + in_flight = self.connection.execute( + "SELECT COUNT(*) FROM pool_reservations " + "WHERE pool_id=? AND reconciled_at IS NULL", + (pool_id,), + ).fetchone()[0] + return derive_pool_capacity( + pool_id, + self._pool_observations(pool_id), + target=target, + in_flight=in_flight, + now=current, + ) + + @staticmethod + def _capacity_target(profile: dict[str, Any]) -> dict[str, str] | None: + fields = ("harness", "model_family", "model_id") + if all(isinstance(profile.get(field), str) and profile[field] for field in fields): + return {field: profile[field] for field in fields} + return None + + def _reserve_pool_capacity_locked( + self, + *, + run_id: str, + pool_id: str, + purpose: str, + profile_id: str | None, + target: dict[str, Any] | None, + max_concurrency: int, + unknown_capacity_policy: str, + frozen_status: str | None, + attempt_id: str | None, + now: datetime, + ) -> dict[str, Any]: + from .capacity import derive_pool_capacity + + if purpose not in {"attempt", "qualification", "classifier"}: + raise ContractError("pool reservation purpose is invalid") + if type(max_concurrency) is not int or max_concurrency < 1: + raise ContractError("pool max_concurrency must be a positive integer") + if unknown_capacity_policy not in {"allow_bounded", "block"}: + raise ContractError("pool unknown_capacity_policy is invalid") + if profile_id is not None and (not isinstance(profile_id, str) or not profile_id): + raise ContractError("pool reservation profile_id is invalid") + if frozen_status is not None and frozen_status not in { + "available", "exhausted", "unknown", + }: + raise ContractError("frozen pool capacity status is invalid") + in_flight = self.connection.execute( + "SELECT COUNT(*) FROM pool_reservations " + "WHERE pool_id=? AND reconciled_at IS NULL", + (pool_id,), + ).fetchone()[0] + snapshot = derive_pool_capacity( + pool_id, + self._pool_observations(pool_id), + target=target, + in_flight=in_flight, + now=now, + ) + live_status = snapshot["status"] + status = "exhausted" if "exhausted" in {frozen_status, live_status} else live_status + if status == "exhausted": + raise ConflictError("account pool capacity is exhausted") + if status == "unknown" and unknown_capacity_policy == "block": + raise ConflictError("unknown account pool capacity is blocked") + effective_concurrency = 1 if status == "unknown" else max_concurrency + if in_flight >= effective_concurrency: + raise ConflictError("account pool concurrency is full") + reservation_id = str(uuid.uuid4()) + self.connection.execute( + "INSERT INTO pool_reservations(" + "id,pool_id,run_id,attempt_id,purpose,profile_id,reserved_at" + ") VALUES(?,?,?,?,?,?,?)", + ( + reservation_id, pool_id, run_id, attempt_id, purpose, profile_id, + now.isoformat(), + ), + ) + return { + "schema_version": 1, + "reservation_id": reservation_id, + "pool_id": pool_id, + "run_id": run_id, + "attempt_id": attempt_id, + "purpose": purpose, + "profile_id": profile_id, + "reserved_at": now.isoformat(), + "capacity_status": status, + "capacity_evidence": snapshot, + } + + def reserve_pool_capacity( + self, + run_id: str, + pool_id: str, + purpose: str, + *, + profile_id: str | None = None, + target: dict[str, Any] | None = None, + max_concurrency: int = 1, + unknown_capacity_policy: str = "allow_bounded", + now: datetime | None = None, + ) -> dict[str, Any]: + """Reserve qualification/classifier capacity under one SQLite write lock.""" + if purpose == "attempt": + raise ContractError("attempt capacity is reserved with its attempt") + current = _authoritative_now(now) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state FROM runs WHERE id=?", (run_id,), + ).fetchone() + if run is None: + raise ContractError("run does not exist") + if run["state"] in TERMINAL_STATES: + raise ConflictError("terminal run cannot reserve pool capacity") + reservation = self._reserve_pool_capacity_locked( + run_id=run_id, + pool_id=pool_id, + purpose=purpose, + profile_id=profile_id, + target=target, + max_concurrency=max_concurrency, + unknown_capacity_policy=unknown_capacity_policy, + frozen_status=None, + attempt_id=None, + now=current, + ) + self.connection.execute("COMMIT") + return reservation + except Exception: + self.connection.execute("ROLLBACK") + raise + + def reconcile_pool_reservation( + self, + reservation_id: str, + reason: str, + *, + now: datetime | None = None, + ) -> dict[str, Any]: + if not isinstance(reservation_id, str) or not reservation_id: + raise ContractError("pool reservation id is invalid") + if not isinstance(reason, str) or not reason: + raise ContractError("pool reconciliation reason is invalid") + current = _authoritative_now(now) + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT pr.*,a.status AS attempt_status FROM pool_reservations pr " + "LEFT JOIN attempts a ON a.id=pr.attempt_id WHERE pr.id=?", + (reservation_id,), + ).fetchone() + if row is None: + raise ContractError("pool reservation does not exist") + if row["reconciled_at"] is not None: + if row["reconcile_reason"] != reason: + raise ConflictError("pool reservation was reconciled differently") + self.connection.execute("COMMIT") + return {**dict(row), "replayed": True} + if row["attempt_id"] is not None and row["attempt_status"] in { + "reserved", "running", "cancelling", "ownership_ambiguous", + }: + raise ConflictError("attempt ownership is not reconciled") + reconciled_at = current.isoformat() + self.connection.execute( + "UPDATE pool_reservations SET reconciled_at=?,reconcile_reason=? " + "WHERE id=? AND reconciled_at IS NULL", + (reconciled_at, reason, reservation_id), + ) + self.connection.execute("COMMIT") + return { + **dict(row), + "reconciled_at": reconciled_at, + "reconcile_reason": reason, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + def active_pool_counts(self) -> dict[str, int]: return { - row["account_pool_id"]: row["in_flight"] + row["pool_id"]: row["in_flight"] for row in self.connection.execute( - "SELECT account_pool_id,COUNT(*) AS in_flight FROM attempts " - "WHERE account_pool_id IS NOT NULL AND status IN " - "('reserved','running','cancelling','ownership_ambiguous') " - "GROUP BY account_pool_id" + "SELECT pool_id,COUNT(*) AS in_flight FROM pool_reservations " + "WHERE reconciled_at IS NULL GROUP BY pool_id" ) } @@ -787,16 +1066,7 @@ def reserve_attempt( if (capacity_status == "unknown" and unknown_policy == "block"): raise ConflictError("unknown account pool capacity is blocked") - effective_concurrency = ( - 1 if capacity_status == "unknown" else max_concurrency - ) - in_flight = self.connection.execute( - "SELECT COUNT(*) FROM attempts WHERE account_pool_id=? " - "AND status IN ('reserved','running','cancelling','ownership_ambiguous')", - (account_pool_id,), - ).fetchone()[0] - if in_flight >= effective_concurrency: - raise ConflictError("account pool concurrency is full") + target = self._capacity_target(selected["profile"]) elif profile_id is not None or profile_index is not None: raise ContractError("attempt profile requires an account pool") old = self.connection.execute("SELECT COALESCE(MAX(fencing_token),0) FROM supervisor_claims WHERE run_id=?", (run_id,)).fetchone()[0] @@ -811,6 +1081,19 @@ def reserve_attempt( attempt_token, now, package_digest, now, role, account_pool_id, profile_id, profile_index), ) + if account_pool_id is not None: + self._reserve_pool_capacity_locked( + run_id=run_id, + pool_id=account_pool_id, + purpose="attempt", + profile_id=profile_id, + target=target, + max_concurrency=max_concurrency, + unknown_capacity_policy=unknown_policy, + frozen_status=capacity_status, + attempt_id=attempt_id, + now=datetime.fromisoformat(now), + ) self.connection.execute("UPDATE runs SET phase='launching',version=?,updated_at=? WHERE id=?", (version, now, run_id)) event = canonical_json({"attempt_id": attempt_id, "supervisor_token": supervisor_token}) self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'supervisor.claimed',?,?)", (run_id, version, event, now)) diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index 31634b8..230a60f 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -1,9 +1,12 @@ import copy +from concurrent.futures import ThreadPoolExecutor from datetime import datetime, timedelta, timezone import math from pathlib import Path +import subprocess import sys import tempfile +import threading import unittest ROOT = Path(__file__).resolve().parents[2] @@ -11,7 +14,7 @@ from devsquad.capacity import derive_pool_capacity, validate_observation from devsquad.contracts import ContractError -from devsquad.store import Store +from devsquad.store import ConflictError, Store NOW = datetime(2026, 9, 27, 15, 0, tzinfo=timezone.utc) @@ -52,6 +55,21 @@ def observation( class CapacityContractTest(unittest.TestCase): + @staticmethod + def make_repository(path): + subprocess.run(["git", "init", "-q", str(path)], check=True) + subprocess.run( + ["git", "-C", str(path), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(path), "config", "user.name", "Test"], + check=True, + ) + (path / "README").write_text("fixture\n") + subprocess.run(["git", "-C", str(path), "add", "README"], check=True) + subprocess.run(["git", "-C", str(path), "commit", "-qm", "base"], check=True) + def test_validation_is_strict_and_normalizes_scopes(self): value = observation(applies_to={ "harnesses": ["grok", "codex"], @@ -174,6 +192,140 @@ def test_migration_nine_creates_capacity_ledger(self): } self.assertTrue({"pool_observations", "pool_reservations"} <= tables) + def test_store_observation_is_replay_safe_and_stale_evidence_stays_visible(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + stale = observation( + "persisted-stale", + expires_at=NOW - timedelta(seconds=1), + used=100, + limit=100, + ) + first = store.record_pool_observation(stale, now=NOW) + replay = store.record_pool_observation(stale, now=NOW) + self.assertFalse(first["replayed"]) + self.assertTrue(replay["replayed"]) + snapshot = store.capacity_snapshot("shared-pool", now=NOW) + self.assertEqual(snapshot["status"], "unknown") + self.assertEqual(snapshot["windows"][0]["reason"], "stale_observation") + + changed = copy.deepcopy(stale) + changed["used"] = 99 + with self.assertRaisesRegex(ConflictError, "different evidence"): + store.record_pool_observation(changed, now=NOW) + + def test_non_attempt_reservation_is_bounded_and_explicitly_reconciled(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + repository = path / "repo" + self.make_repository(repository) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + claim = store.claim_start(repository, "classifier-capacity", {}, "owner") + reserved = store.reserve_pool_capacity( + claim.run_id, "shared-pool", "classifier", now=NOW, + ) + self.assertEqual(store.active_pool_counts(), {"shared-pool": 1}) + with self.assertRaisesRegex(ConflictError, "concurrency"): + store.reserve_pool_capacity( + claim.run_id, "shared-pool", "classifier", now=NOW, + ) + reconciled = store.reconcile_pool_reservation( + reserved["reservation_id"], "classifier_finished", now=NOW, + ) + self.assertFalse(reconciled["replayed"]) + self.assertEqual(store.active_pool_counts(), {}) + replay = store.reconcile_pool_reservation( + reserved["reservation_id"], "classifier_finished", now=NOW, + ) + self.assertTrue(replay["replayed"]) + + def test_two_projects_racing_for_unknown_pool_create_one_reservation(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + repositories = [path / "repo-a", path / "repo-b"] + for repository in repositories: + self.make_repository(repository) + database, artifacts = path / "state.sqlite3", path / "artifacts" + store = Store(database, artifacts) + run_ids = [ + store.claim_start(repository, f"run-{index}", {}, "owner").run_id + for index, repository in enumerate(repositories) + ] + store.close() + barrier = threading.Barrier(2) + + def reserve(run_id): + connection = Store(database, artifacts) + try: + barrier.wait() + return connection.reserve_pool_capacity( + run_id, "shared-pool", "qualification", now=NOW, + )["reservation_id"] + except ConflictError: + return "conflict" + finally: + connection.close() + + with ThreadPoolExecutor(max_workers=2) as executor: + results = list(executor.map(reserve, run_ids)) + self.assertEqual(results.count("conflict"), 1) + winner = next(result for result in results if result != "conflict") + store = Store(database, artifacts) + self.addCleanup(store.close) + self.assertEqual(store.active_pool_counts(), {"shared-pool": 1}) + store.reconcile_pool_reservation(winner, "qualification_finished", now=NOW) + + def test_schema_eight_active_attempt_is_backfilled_and_reconciled(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + database = path / "state.sqlite3" + import sqlite3 + + connection = sqlite3.connect(database) + migrations = ROOT / "plugin/core/src/devsquad/migrations" + for version in range(1, 9): + name = next(migrations.glob(f"{version:03d}_*.sql")) + connection.executescript(name.read_text()) + connection.execute( + "INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)", + (version, NOW.isoformat()), + ) + connection.execute( + "INSERT INTO projects(id,git_common_dir,created_at) VALUES('p','/tmp/p',?)", + (NOW.isoformat(),), + ) + connection.execute( + "INSERT INTO runs(id,project_id,idempotency_key,request_hash," + "submitted_request,state,version,created_at,updated_at,worktree_path) " + "VALUES('r','p','key','hash','{}','running',1,?,?, '/tmp/w')", + (NOW.isoformat(), NOW.isoformat()), + ) + connection.execute( + "INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token," + "status,heartbeat_at,package_digest,created_at,role,account_pool_id,profile_id) " + "VALUES('a','r','p','/tmp/w','token','running',?,'package',?," + "'reviewer','shared-pool','profile-a')", + (NOW.isoformat(), NOW.isoformat()), + ) + connection.commit() + connection.close() + + store = Store(database, path / "artifacts") + self.addCleanup(store.close) + self.assertEqual(store.active_pool_counts(), {"shared-pool": 1}) + row = store.connection.execute( + "SELECT purpose,profile_id FROM pool_reservations WHERE attempt_id='a'", + ).fetchone() + self.assertEqual(tuple(row), ("attempt", "profile-a")) + store.connection.execute( + "UPDATE attempts SET status='finished',finished_at=? WHERE id='a'", + (NOW.isoformat(),), + ) + self.assertEqual(store.active_pool_counts(), {}) + if __name__ == "__main__": unittest.main() diff --git a/test/core/test_store.py b/test/core/test_store.py index 298685a..94148a5 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -295,6 +295,12 @@ def test_shared_pool_reservation_is_transactional_and_releases_on_finish(self): account_pool_id="shared-pool", ) self.assertEqual(self.store.active_pool_counts(), {"shared-pool": 1}) + pool_row = self.store.connection.execute( + "SELECT attempt_id,reconciled_at FROM pool_reservations WHERE run_id=?", + (first.run_id,), + ).fetchone() + self.assertEqual(pool_row["attempt_id"], reservation.attempt_id) + self.assertIsNone(pool_row["reconciled_at"]) with self.assertRaisesRegex(ConflictError, "account pool concurrency"): self.store.reserve_attempt( second.run_id, @@ -309,6 +315,13 @@ def test_shared_pool_reservation_is_transactional_and_releases_on_finish(self): first.run_id, reservation.attempt_token, "succeeded", {}, ) self.assertEqual(self.store.active_pool_counts(), {}) + pool_row = self.store.connection.execute( + "SELECT reconciled_at,reconcile_reason FROM pool_reservations " + "WHERE attempt_id=?", + (reservation.attempt_id,), + ).fetchone() + self.assertIsNotNone(pool_row["reconciled_at"]) + self.assertEqual(pool_row["reconcile_reason"], "attempt_status_finished") released = self.store.reserve_attempt( second.run_id, second_version, @@ -319,6 +332,63 @@ def test_shared_pool_reservation_is_transactional_and_releases_on_finish(self): ) self.assertEqual(released.run_id, second.run_id) + def test_post_preflight_exhaustion_is_rederived_before_reservation(self): + claim = self.store.claim_start( + self.repo, "capacity-changed", {"task": {"budget": {"wall_seconds": 300}}}, + "owner", + ) + version = self.store.complete_preparation( + claim.run_id, claim.fencing_token, self.routed_snapshot(status="available"), + ) + now = datetime.now(timezone.utc) + self.store.record_pool_observation({ + "schema_version": 1, + "observation_id": "weekly-exhausted-after-preflight", + "pool_id": "shared-pool", + "window_id": "weekly", + "applies_to": { + "harnesses": [], "model_families": [], "model_ids": [], + }, + "observed_at": now.isoformat(), + "expires_at": (now + timedelta(minutes=10)).isoformat(), + "source": "native_reported", + "used": 100, + "limit": 100, + "unit": "percent", + "resets_at": (now + timedelta(days=1)).isoformat(), + "confidence": "confirmed", + }, now=now) + with self.assertRaisesRegex(ConflictError, "capacity is exhausted"): + self.store.reserve_attempt( + claim.run_id, version, "supervisor", "package", "reviewer", + account_pool_id="shared-pool", + ) + self.assertEqual(self.store.active_pool_counts(), {}) + + def test_ambiguous_attempt_keeps_pool_until_ownership_is_reconciled(self): + claim = self.store.claim_start( + self.repo, "ambiguous-capacity", {"task": {"budget": {"wall_seconds": 300}}}, + "owner", + ) + version = self.store.complete_preparation( + claim.run_id, claim.fencing_token, self.routed_snapshot(), + ) + reservation = self.store.reserve_attempt( + claim.run_id, version, "supervisor", "package", "reviewer", + account_pool_id="shared-pool", + ) + self.store.mark_attempt_running(reservation, 1001, 1001, "fixture-process") + self.store.block_recovery( + claim.run_id, reservation.attempt_token, "identity ambiguous", + ) + self.assertEqual(self.store.active_pool_counts(), {"shared-pool": 1}) + self.store.request_recovery_cancel(claim.run_id, reservation.attempt_token) + self.assertEqual(self.store.active_pool_counts(), {"shared-pool": 1}) + self.store.finish_recovery_cancel( + claim.run_id, reservation.attempt_token, "absence confirmed", + ) + self.assertEqual(self.store.active_pool_counts(), {}) + def test_unknown_pool_allows_only_one_transactional_trial(self): linked = self.root / "unknown-pool-linked" subprocess.run( From d170c019048e80c026a12a1ed22a06d053ea7779 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 27 Sep 2026 08:41:29 -0700 Subject: [PATCH 097/197] feat: route with persisted capacity evidence --- plugin/core/src/devsquad/cli.py | 31 ++++++- plugin/core/src/devsquad/router.py | 126 ++++++++++++++++++++++++++-- plugin/core/src/devsquad/service.py | 52 ++++++++++-- plugin/core/src/devsquad/store.py | 5 +- test/core/test_cli.py | 31 +++++++ test/core/test_review_runtime.py | 11 ++- test/core/test_router.py | 47 +++++++++++ test/core/test_service.py | 39 +++++++++ 8 files changed, 323 insertions(+), 19 deletions(-) diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 4002064..ba47ce9 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -73,9 +73,24 @@ def command_setup(args: argparse.Namespace) -> tuple[dict, int]: def _read_json(path: str, label: str) -> Any: + def object_pairs(pairs): + value = {} + for key, item in pairs: + if key in value: + raise ValueError(f"duplicate key: {key}") + value[key] = item + return value + + def reject_constant(value): + raise ValueError(f"non-finite number: {value}") + try: - return json.loads(Path(path).read_text()) - except (OSError, UnicodeError, json.JSONDecodeError) as exc: + return json.loads( + Path(path).read_text(), + object_pairs_hook=object_pairs, + parse_constant=reject_constant, + ) + except (OSError, UnicodeError, json.JSONDecodeError, ValueError) as exc: raise ContractError(f"cannot read {label}: {exc}") from exc @@ -144,6 +159,11 @@ def command_start(args: argparse.Namespace) -> tuple[dict, int]: }), 130 +def command_capacity_observe(args: argparse.Namespace) -> tuple[dict, int]: + observation = _read_json(args.file, "capacity observation file") + return envelope(data=_service(args).capacity_observe(observation)), 0 + + def command_status(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).status(args.run)), 0 def command_events(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).events(args.run, args.after, args.limit)), 0 def command_result(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).result(args.run)), 0 @@ -233,6 +253,13 @@ def parser() -> argparse.ArgumentParser: if name == "resume": cmd.add_argument("--recovery-file") cmd.set_defaults(func=fn) events=sub.add_parser("events"); events.add_argument("run"); events.add_argument("--after",type=int,default=0); events.add_argument("--limit",type=int,default=100); events.add_argument("--json",action="store_true"); events.add_argument("--runtime-dir",default=runtime_default); events.set_defaults(func=command_events) + capacity = sub.add_parser("capacity") + capacity_sub = capacity.add_subparsers(dest="capacity_command", required=True) + observe = capacity_sub.add_parser("observe") + observe.add_argument("--file", required=True) + observe.add_argument("--json", action="store_true") + observe.add_argument("--runtime-dir", default=runtime_default) + observe.set_defaults(func=command_capacity_observe) handoff = sub.add_parser("handoff") handoff_sub = handoff.add_subparsers(dest="handoff_command", required=True) claim = handoff_sub.add_parser("claim") diff --git a/plugin/core/src/devsquad/router.py b/plugin/core/src/devsquad/router.py index 7155046..1ac5af0 100644 --- a/plugin/core/src/devsquad/router.py +++ b/plugin/core/src/devsquad/router.py @@ -5,7 +5,7 @@ from datetime import datetime import hashlib import json -from typing import Any +from typing import Any, Callable from .contracts import ( CapabilityUnavailable, @@ -29,6 +29,10 @@ "issue-delivery": ("implementer", "reviewer"), } CAPACITY_STATES = {"available", "exhausted", "unknown"} +CAPACITY_EVIDENCE_FIELDS = { + "schema_version", "pool_id", "status", "in_flight", "observed_at", + "evaluated_at", "target", "windows", "reasons", +} def _strict_json(payload: bytes | str, label: str) -> tuple[dict[str, Any], str]: @@ -63,8 +67,52 @@ def reject_constant(value: str) -> None: return value, hashlib.sha256(encoded).hexdigest() +def _capacity_timestamp(value: Any, field: str, *, nullable: bool = False) -> None: + if value is None and nullable: + return + if not isinstance(value, str) or not value: + raise ContractError(f"capacity {field} must be a timestamp") + try: + parsed = datetime.fromisoformat(value) + except ValueError as exc: + raise ContractError(f"capacity {field} must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError(f"capacity {field} must include a timezone") + + +def _capacity_evidence( + value: Any, + *, + pool_id: str, + target: dict[str, str] | None, +) -> dict[str, Any]: + if not isinstance(value, dict) or set(value) != CAPACITY_EVIDENCE_FIELDS: + raise ContractError("capacity evidence fields are invalid") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("capacity evidence schema_version is invalid") + if value["pool_id"] != pool_id or value["target"] != target: + raise ContractError("capacity evidence identity is invalid") + if not isinstance(value["status"], str) or value["status"] not in CAPACITY_STATES: + raise ContractError("capacity status is invalid") + if type(value["in_flight"]) is not int or value["in_flight"] < 0: + raise ContractError("capacity in_flight must be a non-negative integer") + _capacity_timestamp(value["observed_at"], "observed_at", nullable=True) + _capacity_timestamp(value["evaluated_at"], "evaluated_at") + if not isinstance(value["windows"], list) or not all( + isinstance(item, dict) for item in value["windows"] + ): + raise ContractError("capacity windows must be an object array") + if not isinstance(value["reasons"], list) or not all( + isinstance(item, str) and item for item in value["reasons"] + ): + raise ContractError("capacity reasons must be a string array") + return json.loads(canonical_json(value)) + + def _availability_snapshot( - policy: dict[str, Any], availability: dict[str, Any] | None, + policy: dict[str, Any], + profiles: dict[str, dict[str, Any]], + availability: dict[str, Any] | None, ) -> dict[str, dict[str, Any]]: supplied = {} if availability is None else availability if not isinstance(supplied, dict): @@ -77,6 +125,41 @@ def _availability_snapshot( observation = supplied.get(pool_id, {"status": "unknown", "in_flight": 0}) if not isinstance(observation, dict): raise ContractError("capacity observation must be an object") + if "profiles" in observation: + expected_profiles = { + profile_id: profile + for profile_id, profile in profiles.items() + if profile["account_pool_id"] == pool_id + } + profile_values = observation["profiles"] + if not isinstance(profile_values, dict): + raise ContractError("capacity profile evidence must be an object") + unknown_profiles = set(profile_values) - set(expected_profiles) + if unknown_profiles: + raise ContractError( + f"capacity references unknown pool profiles: {sorted(unknown_profiles)}", + ) + root = _capacity_evidence( + {key: value for key, value in observation.items() if key != "profiles"}, + pool_id=pool_id, + target=None, + ) + root["profiles"] = {} + for profile_id, evidence in profile_values.items(): + profile = expected_profiles[profile_id] + target = { + field: profile[field] + for field in ("harness", "model_family", "model_id") + } + root["profiles"][profile_id] = _capacity_evidence( + evidence, pool_id=pool_id, target=target, + ) + root["max_concurrency"] = pool_policy["max_concurrency"] + root["unknown_capacity_policy"] = pool_policy.get( + "unknown_capacity_policy", "allow_bounded", + ) + result[pool_id] = root + continue unknown = set(observation) - {"status", "in_flight", "observed_at"} missing = {"status", "in_flight"} - set(observation) if unknown or missing: @@ -163,14 +246,15 @@ def _capacity_reason( profile: dict[str, Any], capacity: dict[str, dict[str, Any]], ) -> str | None: pool = capacity[profile["account_pool_id"]] - if pool["status"] == "exhausted": + profile_capacity = pool.get("profiles", {}).get(profile["id"], pool) + if profile_capacity["status"] == "exhausted": return "account_pool_exhausted" - if pool["in_flight"] >= pool["max_concurrency"]: + if profile_capacity["in_flight"] >= pool["max_concurrency"]: return "account_pool_concurrency_full" - if pool["status"] == "unknown": + if profile_capacity["status"] == "unknown": if pool["unknown_capacity_policy"] == "block": return "unknown_capacity_blocked" - if pool["in_flight"] >= 1: + if profile_capacity["in_flight"] >= 1: return "unknown_capacity_trial_in_flight" return None @@ -256,7 +340,7 @@ def resolve_routing( validate_policy(policy) profiles = {profile["id"]: profile for profile in profile_registry["profiles"]} bindings = profile_registry["bindings"] - capacity = _availability_snapshot(policy, availability) + capacity = _availability_snapshot(policy, profiles, availability) minimum_quality = policy["task_classes"].get(task["task_class"]) if minimum_quality is None: raise PolicyDenied(f"policy does not authorize task class: {task['task_class']}") @@ -411,3 +495,31 @@ def capacity_with_live_reservations( } for pool_id in policy["account_pools"] } + + +def capacity_with_saved_observations( + profiles_payload: bytes | str, + policy_payload: bytes | str, + snapshot: Callable[..., dict[str, Any]], +) -> dict[str, dict[str, Any]]: + """Build per-profile capacity evidence from the shared persisted ledger.""" + registry, _ = _strict_json(profiles_payload, "profiles file") + policy, _ = _strict_json(policy_payload, "policy file") + validate_profile_registry(registry) + validate_policy(policy) + if not callable(snapshot): + raise ContractError("capacity snapshot provider is invalid") + result: dict[str, dict[str, Any]] = {} + for pool_id in policy["account_pools"]: + root = snapshot(pool_id, target=None) + profile_evidence = {} + for profile in registry["profiles"]: + if profile["account_pool_id"] != pool_id: + continue + target = { + field: profile[field] + for field in ("harness", "model_family", "model_id") + } + profile_evidence[profile["id"]] = snapshot(pool_id, target=target) + result[pool_id] = {**root, "profiles": profile_evidence} + return result diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 3147557..e818230 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -28,7 +28,7 @@ project_branch_review_history, validate_saved_review_handoff, ) -from .router import capacity_with_live_reservations, load_routing +from .router import capacity_with_saved_observations, load_routing from .store import ( ConflictError, HandoffClaim, @@ -424,7 +424,7 @@ def _resolve_snapshot( internal_review_fixture: dict[str, Any] | None = None, internal_lead_fixture: dict[str, Any] | None = None, internal_implementation_fixture: dict[str, Any] | None = None, - capacity_in_flight: dict[str, int] | None = None, + capacity_store: Store | None = None, ) -> dict[str, Any]: repo = resolved_repo or Path(task["project"]["repo_path"]).resolve(strict=True) base_oid = resolve_commit(repo, task["project"]["base_ref"]) @@ -460,12 +460,16 @@ def _resolve_snapshot( if internal_delay < 0 or internal_delay > 60: raise ContractError("internal fake delay is invalid") snapshot["internal_fake_delay"] = internal_delay else: + if capacity_store is None: + raise ContractError("public preflight requires shared capacity state") snapshot["routing"] = load_routing( task, config_payloads["profiles_file"], config_payloads["policy_file"], - availability=capacity_with_live_reservations( - config_payloads["policy_file"], capacity_in_flight or {}, + availability=capacity_with_saved_observations( + config_payloads["profiles_file"], + config_payloads["policy_file"], + capacity_store.capacity_snapshot, ), ) if project_id is None or run_id is None: @@ -618,7 +622,7 @@ def _continue_preparation( internal_review_fixture=internal_review_fixture, internal_lead_fixture=internal_lead_fixture, internal_implementation_fixture=internal_implementation_fixture, - capacity_in_flight=store.active_pool_counts(), + capacity_store=store, ) if (task["workflow"] == "issue-delivery" and internal_delay is None @@ -774,6 +778,43 @@ def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, diges ).start() return process.pid + def capacity_observe(self, observation: dict[str, Any]) -> dict[str, Any]: + """Record one capacity observation and return its current pool view.""" + store = self._store() + try: + recorded = store.record_pool_observation(observation) + return { + "record": recorded, + "capacity": store.capacity_snapshot(observation.get("pool_id")), + } + finally: + store.close() + + @staticmethod + def _status_capacity(store: Store, run: dict[str, Any]) -> dict[str, Any] | None: + try: + snapshot = json.loads(run["mutable_snapshot"]) + routing = snapshot["routing"] + roles = routing["roles"] + frozen = routing["capacity"] + except (KeyError, TypeError, json.JSONDecodeError): + return None + current = {} + for role in roles.values(): + for candidate in [role["selected"], *role.get("fallbacks", [])]: + profile = candidate["profile"] + profile_id = candidate["profile_id"] + if profile_id in current: + continue + target = { + field: profile[field] + for field in ("harness", "model_family", "model_id") + } + current[profile_id] = store.capacity_snapshot( + profile["account_pool_id"], target=target, + ) + return {"frozen": frozen, "current": current} + def status(self, run_id: str) -> dict[str, Any]: store = self._store() try: @@ -814,6 +855,7 @@ def status(self, run_id: str) -> dict[str, Any]: } if active else None, "handoff": self._handoff_payload(handoff, include_packet=False) if handoff else None, "next_action": next_action, + "capacity": self._status_capacity(store, run), } finally: store.close() diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index a2fe1b3..418890d 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1039,9 +1039,12 @@ def reserve_attempt( "attempt profile does not match frozen routing order" ) capacity = snapshot["routing"]["capacity"][account_pool_id] + profile_capacity = capacity.get("profiles", {}).get( + selected.get("profile_id"), capacity, + ) expected_pool = selected["profile"]["account_pool_id"] max_concurrency = capacity["max_concurrency"] - capacity_status = capacity.get("status", "available") + capacity_status = profile_capacity.get("status", "available") unknown_policy = capacity.get( "unknown_capacity_policy", "allow_bounded", ) diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 45b96ee..225df6c 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -103,6 +103,28 @@ def test_status_events_result_cancel_and_resume_operations(self): self.assert_success_envelope(payload, response) service.resume.assert_called_once_with("run-1", {"attempt_id": "a-1", "disposition": "confirm_dead"}) + def test_capacity_observe_dispatches_file(self): + observation = { + "schema_version": 1, + "observation_id": "obs-1", + "pool_id": "pool-a", + } + observation_file = self.root / "capacity.json" + observation_file.write_text(json.dumps(observation)) + service = mock.Mock() + response = { + "record": {"replayed": False}, + "capacity": {"status": "unknown"}, + } + service.capacity_observe.return_value = response + code, payload, stderr = self.invoke([ + "capacity", "observe", "--file", str(observation_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.capacity_observe.assert_called_once_with(observation) + def test_handoff_claim_renew_and_complete_dispatch_parsed_objects(self): claim_payload = { "schema_version": 1, @@ -179,6 +201,15 @@ def test_parser_and_json_file_failures_are_input_errors(self): self.assertEqual(payload["error"]["code"], "INPUT_INVALID") self.assertIn("cannot read task file", payload["error"]["message"]) + malformed.write_text('{"schema_version":1,"schema_version":1}') + code, payload, _ = self.invoke([ + "start", "--task-file", str(malformed), + "--idempotency-key", "duplicate-json", + "--runtime-dir", str(self.runtime), "--json", + ]) + self.assertEqual(code, 64) + self.assertIn("duplicate key", payload["error"]["message"]) + def test_setup_filters_hosts_and_treats_a_valid_dry_run_as_completed(self): codex = mock.Mock(id="codex") grok = mock.Mock(id="grok") diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index 62e2e54..4567fb9 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -14,6 +14,7 @@ ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.capacity import derive_pool_capacity from devsquad.contracts import ContractError from devsquad.reports import TERMINAL_REPORT_NAMES, build_handoff_reports from devsquad.service import Service @@ -842,10 +843,12 @@ def test_cancel_waiting_review_publishes_full_terminal_reports(self): self.assertEqual(len(receipt["attempts"]), 1) def test_public_preflight_observes_live_shared_pool_reservations(self): - with patch.object( - Store, "active_pool_counts", - return_value={"fixture-subscription": 1}, - ): + def full_unknown_pool(_store, pool_id, *, target=None, now=None): + return derive_pool_capacity( + pool_id, [], target=target, in_flight=1, now=now, + ) + + with patch.object(Store, "capacity_snapshot", full_unknown_pool): started = self.service.start( self.task, "pool-full-before-launch", diff --git a/test/core/test_router.py b/test/core/test_router.py index 2674553..05435ab 100644 --- a/test/core/test_router.py +++ b/test/core/test_router.py @@ -1,4 +1,5 @@ import copy +from datetime import datetime, timezone import hashlib import json from pathlib import Path @@ -302,6 +303,52 @@ def test_typed_unknown_and_concurrency_capacity_are_fail_closed(self): availability={"pool-a": {"status": "available", "in_flight": 3}}, ) + def test_profile_scoped_windows_exclude_only_applicable_candidate(self): + now = datetime(2026, 9, 27, 16, 0, tzinfo=timezone.utc).isoformat() + + def evidence(profile, status, reason): + target = None if profile is None else { + "harness": profile["harness"], + "model_family": profile["model_family"], + "model_id": profile["model_id"], + } + return { + "schema_version": 1, + "pool_id": "pool-a", + "status": status, + "in_flight": 0, + "observed_at": now, + "evaluated_at": now, + "target": target, + "windows": [{"reason": reason}], + "reasons": [] if status == "available" else [f"{reason}:weekly"], + } + + review_a, review_b = self.registry["profiles"] + availability = { + "pool-a": { + **evidence(None, "unknown", "no_applicable_observations"), + "profiles": { + "review-a": evidence(review_a, "available", "window_available"), + "review-b": evidence(review_b, "exhausted", "window_exhausted"), + }, + }, + } + routed = resolve_routing( + self.task, self.registry, self.policy, availability=availability, + ) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "review-a") + self.assertEqual(reviewer["excluded"][0], { + "reference": {"kind": "alias", "id": "review.deep"}, + "profile_id": "review-b", + "reason": "account_pool_exhausted", + }) + self.assertEqual( + routed["capacity"]["pool-a"]["profiles"]["review-b"]["reasons"], + ["window_exhausted:weekly"], + ) + def test_policy_missing_task_class_or_required_role_is_denied(self): del self.policy["task_classes"]["fixture-review"] with self.assertRaisesRegex(PolicyDenied, "does not authorize task class"): diff --git a/test/core/test_service.py b/test/core/test_service.py index 3e0d7c0..baace58 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -10,6 +10,7 @@ import time import unittest from unittest import mock +from datetime import datetime, timedelta, timezone ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "plugin/core/src")) @@ -126,6 +127,44 @@ def test_start_is_idempotent_and_result_events_are_durable(self): self.assertEqual(len(page["events"]),2); self.assertIsNotNone(page["next_cursor"]) with self.assertRaises(ConflictError): self.service.resume(first["run_id"]) + def test_capacity_observation_drives_preflight_and_status_evidence(self): + now = datetime.now(timezone.utc) + observed = self.service.capacity_observe({ + "schema_version": 1, + "observation_id": "service-capacity-short", + "pool_id": "fixture-subscription", + "window_id": "short", + "applies_to": { + "harnesses": [], "model_families": [], "model_ids": [], + }, + "observed_at": now.isoformat(), + "expires_at": (now + timedelta(minutes=10)).isoformat(), + "source": "manual_reported", + "used": 10, + "limit": 100, + "unit": "percent", + "resets_at": (now + timedelta(hours=1)).isoformat(), + "confidence": "reported", + }) + self.assertFalse(observed["record"]["replayed"]) + self.assertEqual(observed["capacity"]["status"], "available") + started = self.service.start( + self.task, + "capacity-status", + _internal_review_fixture={ + "verdict": "clean", "summary": "capacity fixture", "findings": [], + }, + ) + self.assertEqual(started["state"], "queued", started) + status = self.wait_state(started["run_id"], {"awaiting_host"}) + self.assertEqual( + status["capacity"]["frozen"]["fixture-subscription"]["status"], + "available", + ) + current = status["capacity"]["current"]["fixture-reviewer"] + self.assertEqual(current["status"], "available") + self.assertEqual(current["windows"][0]["window_id"], "short") + def test_abandoned_preparation_is_reclaimed_from_the_submitted_request(self): store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") try: From 3e3e9520212279e7219a260c160d02034173ba35 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 27 Sep 2026 08:42:43 -0700 Subject: [PATCH 098/197] docs: checkpoint M6 shared capacity --- docs/plans/engineering-team/M6-STATUS.md | 51 ++++++++++++++++++++++++ docs/plans/engineering-team/RESUME.md | 23 +++++++---- docs/plans/engineering-team/backlog.json | 11 ++++- 3 files changed, 77 insertions(+), 8 deletions(-) create mode 100644 docs/plans/engineering-team/M6-STATUS.md diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md new file mode 100644 index 0000000..fe4eccd --- /dev/null +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -0,0 +1,51 @@ +# M6 implementation status + +M6 is **in progress**. Shared capacity observations, reservations and routing +are verified; outcomes, experiments, lifecycle promotion/rollback and the +default-off decision helper remain open. + +| Requirement | Planned evidence | Status | +|---|---|---| +| Strict capacity evidence | Typed pool/window/scope/source/confidence/TTL validation; stale, estimated and incomplete measurements remain unknown | verified at `21b1331` | +| Shared transactional reservations | Two projects share one pool; one-slot races produce one owner; schema-8 active attempts survive migration; ambiguous ownership retains the reservation | verified at `26ce5cf` | +| Capacity-aware routing | All applicable windows and profile sublimits affect deterministic selection; a later exhausted observation blocks reservation; paid API remains policy-gated | verified at `d170c01` | +| Public observation/status surface | `squad capacity observe --file FILE`; replay-safe persistence; status shows frozen and current detailed evidence | verified at `d170c01` | +| Final and late outcomes | Preserve attempt contribution, lead repair, final success and escaped-defect corrections without crediting failed attempts | pending | +| Comparison reports and proposals | Sample sizes, missingness, selection mode, one-variable experiment, held-out rerun, no-change/promotion proposal and rollback target | pending | +| Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | pending | +| Decision helper M6-D1 | Default-off typed contract, fake adapter, cache/accounting and authority/integrity tests | in progress; synthetic Jev probe mechanics only | +| Jev M6-D2 | One capped synthetic request with exact model/usage/latency/cost receipt | blocked on `TYPESAFE_API_KEY` | +| Laya M6-D3 | Triggered pinned local comparison and measured keep-off/adopt decision | pending; run only if the declared Jev trigger fires | + +## Capacity checkpoint + +Schema 9 stores immutable capacity observations and explicit reservations. +Each observation retains its native unit and applicability scope; the latest +observation per scoped window is evaluated without inventing quota. A fresh +authoritative exhausted window wins over shorter available windows. Stale, +estimated, incomplete or absent evidence remains `unknown`, and `allow_bounded` +permits only one unresolved reservation. + +Reservations use the shared SQLite write fence across projects. Attempt +reservations are created atomically with the attempt and reconciled by the same +transaction that proves ownership released; `ownership_ambiguous` remains in +flight. Migration backfills active schema-8 attempts so an update cannot create +a duplicate allowance. + +Public preflight freezes detailed pool and per-profile evidence. Reservation +rederives current observations under the transaction, so evidence that changes +after preflight cannot launch against an exhausted pool. Profile/model-family +sublimits exclude only applicable candidates. The public CLI records strict, +replay-safe JSON, and run status returns both frozen and current windows. + +The checkpoint gate is **255 core tests with 2 optional-SDK skips** and +ResourceWarning promoted to error, plus **220/220 Bash assertions**. + +## Exact next slice + +Add the schema-10 outcome ledger and `learning.py`: strict final/late records, +append-only corrections, attempt/lead contribution attribution, selection-mode +separation and comparison reports with sample sizes and missingness. Prove that +a failed original attempt later repaired by another profile yields final task +success without crediting the failed attempt, and that a late escaped defect +updates history without erasing the original verdict. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 53fa4e6..4c9d6b6 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -131,6 +131,16 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne error; the compatibility gate is 220 Bash assertions. M5 is now blocked only on normal Claude login for the genuine Claude implementation → different- model Codex review/check/disposition receipt. +- M6 shared-capacity work is verified through `d170c01`. Schema 9 persists + strict scoped observations and reservations, backfills active schema-8 + attempts and keeps ambiguous ownership in flight. Two separate projects + racing for an unresolved pool create one reservation; fresh exhausted weekly + evidence beats short-window availability; stale/estimated/incomplete data + remains unknown; reservation rechecks post-preflight changes; and routing + applies model/profile sublimits without widening eligibility. `squad capacity + observe --file FILE` and frozen/current status evidence are wired. The gate + is 255 core tests with 2 optional-SDK skips and 220 Bash assertions. See + [M6-STATUS.md](M6-STATUS.md). - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -157,7 +167,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 241 discovered through M5 live-writer recovery; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | +| Python core discovery | 255 discovered through M6 capacity routing; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -194,12 +204,11 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Begin M6 with the shared-capacity observation/reservation model: persist all - applicable windows, derive available/exhausted/unknown from fresh evidence, - fence concurrent reservations, expose the snapshot through the service/CLI - and feed it into deterministic routing. Then add evidence-based outcomes, - lifecycle qualification/guarded promotion and the default-off decision - helper. Do not rebuild the completed M5 offline path. +2. Continue M6 with the outcome/learning ledger: final and late corrections, + attempt/lead contribution, selection-mode separation, comparable reports + and one-variable proposal/rollback evidence. Then add lifecycle + qualification/guarded promotion and the default-off decision helper. Do not + rebuild the completed M5 offline path or capacity ledger. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index ca97de3..794cd4a 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -275,7 +275,7 @@ "id": "M6", "title": "Shared capacity and evidence-based improvement", "depends_on": ["M4", "M5"], - "status": "pending", + "status": "in_progress", "acceptance_section": "M6 — Make capacity and improvement evidence useful", "decision_helper_work_packages": [ { @@ -316,6 +316,15 @@ } ], "evidence": [ + { + "kind": "capacity_checkpoint", + "revision": "d170c01", + "command_or_action": "255 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, two-project reservation race, schema-8 active-attempt migration and public capacity CLI/service/routing tests", + "outcome": "Schema 9 persists strict scoped quota-window observations and reservations; fresh exhaustion, stale/unknown evidence, model-family sublimits, post-preflight changes and ambiguous ownership now affect deterministic routing and transactional launch without widening billing or permission eligibility.", + "artifact": "M6-STATUS.md", + "recorded_at": "2026-09-27T08:41:45-07:00", + "availability": "tracked_tests" + }, { "kind": "early_experiment_checkpoint", "revision": "70e59cb", From d621df2f9d41b5a11a76c42f98e4989c27f0717c Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 27 Sep 2026 08:54:08 -0700 Subject: [PATCH 099/197] feat: record final and late outcomes --- plugin/core/src/devsquad/learning.py | 193 ++++++++++++++++++ .../src/devsquad/migrations/010_outcomes.sql | 26 +++ plugin/core/src/devsquad/store.py | 154 +++++++++++++- test/core/test_capacity.py | 2 +- test/core/test_cli.py | 4 +- test/core/test_handoff_store.py | 6 +- test/core/test_learning.py | 183 +++++++++++++++++ test/core/test_store.py | 8 +- 8 files changed, 564 insertions(+), 12 deletions(-) create mode 100644 plugin/core/src/devsquad/learning.py create mode 100644 plugin/core/src/devsquad/migrations/010_outcomes.sql create mode 100644 test/core/test_learning.py diff --git a/plugin/core/src/devsquad/learning.py b/plugin/core/src/devsquad/learning.py new file mode 100644 index 0000000..9f67a19 --- /dev/null +++ b/plugin/core/src/devsquad/learning.py @@ -0,0 +1,193 @@ +"""Strict outcome evidence and comparison primitives for M6 learning.""" + +from __future__ import annotations + +from datetime import datetime, timedelta, timezone +import json +from typing import Any + +from .contracts import ContractError +from .store import canonical_json + + +OUTCOME_FIELDS = { + "schema_version", "outcome_id", "kind", "verdict", "selection_mode", + "observed_at", "corrects_outcome_id", "summary", "criteria", + "contributions", "lead_repairs", "evidence_refs", +} +FINAL_VERDICTS = {"succeeded", "failed", "cancelled"} +LATE_VERDICTS = {"escaped_defect", "corrected"} +SELECTION_MODES = {"automatic", "pinned", "experimental"} +CRITERION_STATES = {"passed", "failed", "unknown"} +CONTRIBUTION_RESULTS = {"failed", "successful", "repair", "finding", "neutral"} +ROLES = {"worker", "implementer", "reviewer", "lead", "researcher"} +MAX_CLOCK_SKEW = timedelta(minutes=5) + + +def _now(value: datetime | None) -> datetime: + current = datetime.now(timezone.utc) if value is None else value + if not isinstance(current, datetime) or current.tzinfo is None or current.utcoffset() is None: + raise ContractError("outcome evaluation time must include a timezone") + return current.astimezone(timezone.utc) + + +def _timestamp(value: Any, field: str) -> datetime: + if not isinstance(value, str) or not value: + raise ContractError(f"outcome {field} must be an ISO timestamp") + try: + parsed = datetime.fromisoformat(value) + except ValueError as exc: + raise ContractError(f"outcome {field} must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError(f"outcome {field} must include a timezone") + return parsed.astimezone(timezone.utc) + + +def _identifier(value: Any, field: str, *, nullable: bool = False) -> str | None: + if value is None and nullable: + return None + if not isinstance(value, str) or not value.strip(): + suffix = " or null" if nullable else "" + raise ContractError(f"outcome {field} must be a non-empty string{suffix}") + return value + + +def _evidence_refs(value: Any, field: str, *, required: bool = False) -> list[str]: + if not isinstance(value, list) or any( + not isinstance(item, str) or not item for item in value + ): + raise ContractError(f"outcome {field} must be a string array") + if len(set(value)) != len(value): + raise ContractError(f"outcome {field} must be unique") + if required and not value: + raise ContractError(f"outcome {field} must not be empty") + return list(value) + + +def validate_outcome( + value: dict[str, Any], *, now: datetime | None = None, +) -> dict[str, Any]: + """Validate and detach one final or late-correction outcome record.""" + if not isinstance(value, dict) or set(value) != OUTCOME_FIELDS: + unknown = sorted(set(value) - OUTCOME_FIELDS) if isinstance(value, dict) else [] + missing = sorted(OUTCOME_FIELDS - set(value)) if isinstance(value, dict) else [] + raise ContractError( + f"outcome fields invalid: unknown={unknown} missing={missing}", + ) + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("outcome schema_version is invalid") + outcome_id = _identifier(value["outcome_id"], "outcome_id") + kind = value["kind"] + if not isinstance(kind, str) or kind not in {"final", "late_correction"}: + raise ContractError("outcome kind is invalid") + verdict = value["verdict"] + allowed_verdicts = FINAL_VERDICTS if kind == "final" else LATE_VERDICTS + if not isinstance(verdict, str) or verdict not in allowed_verdicts: + raise ContractError("outcome verdict is invalid for its kind") + selection_mode = value["selection_mode"] + if not isinstance(selection_mode, str) or selection_mode not in SELECTION_MODES: + raise ContractError("outcome selection_mode is invalid") + corrects = _identifier( + value["corrects_outcome_id"], "corrects_outcome_id", nullable=True, + ) + if (kind == "final" and corrects is not None) or ( + kind == "late_correction" and corrects is None + ): + raise ContractError("outcome correction reference is invalid") + if not isinstance(value["summary"], str) or not value["summary"].strip(): + raise ContractError("outcome summary must be a non-empty string") + observed = _timestamp(value["observed_at"], "observed_at") + if observed > _now(now) + MAX_CLOCK_SKEW: + raise ContractError("outcome observed_at exceeds allowed clock skew") + + if not isinstance(value["criteria"], list): + raise ContractError("outcome criteria must be an array") + criteria = [] + criterion_ids = set() + for item in value["criteria"]: + if not isinstance(item, dict) or set(item) != { + "criterion_id", "status", "evidence_refs", + }: + raise ContractError("outcome criterion fields are invalid") + criterion_id = _identifier(item["criterion_id"], "criterion_id") + if criterion_id in criterion_ids: + raise ContractError("outcome criterion ids must be unique") + criterion_ids.add(criterion_id) + if not isinstance(item["status"], str) or item["status"] not in CRITERION_STATES: + raise ContractError("outcome criterion status is invalid") + criteria.append({ + "criterion_id": criterion_id, + "status": item["status"], + "evidence_refs": _evidence_refs( + item["evidence_refs"], "criterion evidence_refs", + ), + }) + if kind == "final" and verdict == "succeeded" and any( + item["status"] != "passed" for item in criteria + ): + raise ContractError("successful outcome criteria must all pass") + + if not isinstance(value["contributions"], list): + raise ContractError("outcome contributions must be an array") + contributions = [] + contribution_attempts = set() + for item in value["contributions"]: + if not isinstance(item, dict) or set(item) != { + "attempt_id", "role", "result", "independent_success", "evidence_refs", + }: + raise ContractError("outcome contribution fields are invalid") + attempt_id = _identifier(item["attempt_id"], "attempt_id") + if attempt_id in contribution_attempts: + raise ContractError("outcome contribution attempt ids must be unique") + contribution_attempts.add(attempt_id) + if not isinstance(item["role"], str) or item["role"] not in ROLES: + raise ContractError("outcome contribution role is invalid") + if not isinstance(item["result"], str) or item["result"] not in CONTRIBUTION_RESULTS: + raise ContractError("outcome contribution result is invalid") + if type(item["independent_success"]) is not bool: + raise ContractError("outcome independent_success must be boolean") + if item["independent_success"] and item["result"] != "successful": + raise ContractError("only a successful contribution can be independently successful") + contributions.append({ + "attempt_id": attempt_id, + "role": item["role"], + "result": item["result"], + "independent_success": item["independent_success"], + "evidence_refs": _evidence_refs( + item["evidence_refs"], "contribution evidence_refs", + ), + }) + + if not isinstance(value["lead_repairs"], list): + raise ContractError("outcome lead_repairs must be an array") + lead_repairs = [] + for item in value["lead_repairs"]: + if not isinstance(item, dict) or set(item) != { + "lead_attempt_id", "description", "evidence_refs", + }: + raise ContractError("outcome lead repair fields are invalid") + lead_attempt_id = _identifier( + item["lead_attempt_id"], "lead_attempt_id", nullable=True, + ) + if not isinstance(item["description"], str) or not item["description"].strip(): + raise ContractError("outcome lead repair description is invalid") + lead_repairs.append({ + "lead_attempt_id": lead_attempt_id, + "description": item["description"], + "evidence_refs": _evidence_refs( + item["evidence_refs"], "lead repair evidence_refs", + ), + }) + + normalized = { + **value, + "outcome_id": outcome_id, + "corrects_outcome_id": corrects, + "criteria": criteria, + "contributions": contributions, + "lead_repairs": lead_repairs, + "evidence_refs": _evidence_refs( + value["evidence_refs"], "evidence_refs", required=True, + ), + } + return json.loads(canonical_json(normalized)) diff --git a/plugin/core/src/devsquad/migrations/010_outcomes.sql b/plugin/core/src/devsquad/migrations/010_outcomes.sql new file mode 100644 index 0000000..fe5bcf9 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/010_outcomes.sql @@ -0,0 +1,26 @@ +CREATE TABLE outcomes ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + outcome_id TEXT NOT NULL UNIQUE, + run_id TEXT NOT NULL REFERENCES runs(id), + kind TEXT NOT NULL CHECK(kind IN ('final', 'late_correction')), + verdict TEXT NOT NULL CHECK(verdict IN ('succeeded', 'failed', 'cancelled', 'escaped_defect', 'corrected')), + selection_mode TEXT NOT NULL CHECK(selection_mode IN ('automatic', 'pinned', 'experimental')), + observed_at TEXT NOT NULL, + corrects_outcome_id TEXT REFERENCES outcomes(outcome_id), + payload_json TEXT NOT NULL, + payload_sha256 TEXT NOT NULL + CHECK(length(payload_sha256) = 64 AND payload_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL, + CHECK( + (kind = 'final' AND verdict IN ('succeeded', 'failed', 'cancelled') AND corrects_outcome_id IS NULL) + OR + (kind = 'late_correction' AND verdict IN ('escaped_defect', 'corrected') AND corrects_outcome_id IS NOT NULL) + ) +); + +CREATE UNIQUE INDEX one_final_outcome_per_run +ON outcomes(run_id) +WHERE kind = 'final'; + +CREATE INDEX outcomes_run_history +ON outcomes(run_id, observed_at, id); diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 418890d..ad09cc7 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 9 +SUPPORTED_SCHEMA_VERSION = 10 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -901,7 +901,7 @@ def reserve_pool_capacity( self.connection.execute("BEGIN IMMEDIATE") try: run = self.connection.execute( - "SELECT state FROM runs WHERE id=?", (run_id,), + "SELECT state,mutable_snapshot FROM runs WHERE id=?", (run_id,), ).fetchone() if run is None: raise ContractError("run does not exist") @@ -981,6 +981,156 @@ def active_pool_counts(self) -> dict[str, int]: ) } + def record_outcome( + self, + run_id: str, + outcome: dict[str, Any], + *, + now: datetime | None = None, + ) -> dict[str, Any]: + """Append one replay-safe final outcome or late correction.""" + from .learning import validate_outcome + + current = _authoritative_now(now) + normalized = validate_outcome(outcome, now=current) + payload = canonical_json(normalized) + digest = hashlib.sha256(payload.encode()).hexdigest() + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT run_id,payload_json,recorded_at FROM outcomes WHERE outcome_id=?", + (normalized["outcome_id"],), + ).fetchone() + if existing is not None: + if existing["run_id"] != run_id or existing["payload_json"] != payload: + raise ConflictError( + "outcome id was already used with different evidence", + ) + self.connection.execute("COMMIT") + return { + "run_id": run_id, + "outcome": normalized, + "recorded_at": existing["recorded_at"], + "replayed": True, + } + run = self.connection.execute( + "SELECT state,mutable_snapshot FROM runs WHERE id=?", (run_id,), + ).fetchone() + if run is None: + raise ContractError("run does not exist") + if run["state"] not in TERMINAL_STATES: + raise ConflictError("outcomes require a terminal run") + try: + snapshot = json.loads(run["mutable_snapshot"] or "null") + roles = snapshot["routing"]["roles"].values() + expected_selection_mode = ( + "experimental" + if snapshot.get("experiment_assignment") is not None + else "pinned" + if any(role.get("source") == "override" for role in roles) + else "automatic" + ) + except (KeyError, TypeError, json.JSONDecodeError): + expected_selection_mode = None + if (expected_selection_mode is not None + and normalized["selection_mode"] != expected_selection_mode): + raise ConflictError("outcome selection mode does not match frozen routing") + if normalized["kind"] == "final": + if normalized["verdict"] != run["state"]: + raise ConflictError("final outcome verdict does not match run state") + else: + corrected = self.connection.execute( + "SELECT run_id,kind,selection_mode,observed_at FROM outcomes " + "WHERE outcome_id=?", + (normalized["corrects_outcome_id"],), + ).fetchone() + if (corrected is None or corrected["run_id"] != run_id + or corrected["kind"] != "final"): + raise ConflictError("late outcome must correct this run's final outcome") + if corrected["selection_mode"] != normalized["selection_mode"]: + raise ConflictError("late outcome selection mode changed") + if datetime.fromisoformat(normalized["observed_at"]) < datetime.fromisoformat( + corrected["observed_at"], + ): + raise ConflictError("late outcome predates the final outcome") + + attempts = { + row["id"]: row + for row in self.connection.execute( + "SELECT id,role,status,output_metadata FROM attempts WHERE run_id=?", + (run_id,), + ) + } + for contribution in normalized["contributions"]: + attempt = attempts.get(contribution["attempt_id"]) + if attempt is None or attempt["role"] != contribution["role"]: + raise ConflictError("outcome contribution does not match run attempt") + if attempt["status"] != "finished": + raise ConflictError("outcome contribution attempt is not finished") + if attempt["output_metadata"]: + try: + metadata = json.loads(attempt["output_metadata"]) + except (TypeError, json.JSONDecodeError) as exc: + raise ConflictError("attempt output metadata is invalid") from exc + if (isinstance(metadata, dict) and metadata.get("failure") is not None + and (contribution["result"] != "failed" + or contribution["independent_success"])): + raise ConflictError( + "failed attempt cannot receive successful contribution credit", + ) + for repair in normalized["lead_repairs"]: + attempt_id = repair["lead_attempt_id"] + if attempt_id is None: + continue + attempt = attempts.get(attempt_id) + if attempt is None or attempt["role"] != "lead" or attempt["status"] != "finished": + raise ConflictError("lead repair does not match a finished lead attempt") + + recorded_at = current.isoformat() + self.connection.execute( + "INSERT INTO outcomes(outcome_id,run_id,kind,verdict,selection_mode," + "observed_at,corrects_outcome_id,payload_json,payload_sha256,recorded_at) " + "VALUES(?,?,?,?,?,?,?,?,?,?)", + ( + normalized["outcome_id"], run_id, normalized["kind"], + normalized["verdict"], normalized["selection_mode"], + normalized["observed_at"], normalized["corrects_outcome_id"], + payload, digest, recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "run_id": run_id, + "outcome": normalized, + "recorded_at": recorded_at, + "replayed": False, + } + except sqlite3.IntegrityError as exc: + self.connection.execute("ROLLBACK") + raise ConflictError("run already has a final outcome") from exc + except Exception: + self.connection.execute("ROLLBACK") + raise + + def outcomes_for_run(self, run_id: str) -> list[dict[str, Any]]: + if not self.connection.execute( + "SELECT 1 FROM runs WHERE id=?", (run_id,), + ).fetchone(): + raise ContractError("run does not exist") + return [ + { + "run_id": row["run_id"], + "outcome": json.loads(row["payload_json"]), + "payload_sha256": row["payload_sha256"], + "recorded_at": row["recorded_at"], + } + for row in self.connection.execute( + "SELECT run_id,payload_json,payload_sha256,recorded_at " + "FROM outcomes WHERE run_id=? ORDER BY observed_at,id", + (run_id,), + ) + ] + def worker_invocations(self, run_id: str) -> int: """Count attempts whose durable runner actually crossed the launch fence.""" return self.connection.execute( diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index 230a60f..87fec4e 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -184,7 +184,7 @@ def test_migration_nine_creates_capacity_ledger(self): version = store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 9) + self.assertEqual(version, 10) tables = { row[0] for row in store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 225df6c..d1c2293 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -387,7 +387,7 @@ def build_python(): return candidate return None - def test_installed_wheel_contains_and_applies_migrations_through_nine(self): + def test_installed_wheel_contains_and_applies_migrations_through_ten(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") @@ -429,7 +429,7 @@ def test_installed_wheel_contains_and_applies_migrations_through_nine(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 9 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 10 attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 286795f..244b0ae 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 9, + 10, ) tables = { row[0] @@ -618,7 +618,7 @@ def build_python(): return candidate return None - def test_installed_wheel_applies_schema_four_to_nine(self): + def test_installed_wheel_applies_schema_four_to_ten(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel") @@ -684,7 +684,7 @@ def test_installed_wheel_applies_schema_four_to_nine(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 9 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 10 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_learning.py b/test/core/test_learning.py new file mode 100644 index 0000000..62ad73e --- /dev/null +++ b/test/core/test_learning.py @@ -0,0 +1,183 @@ +import copy +from datetime import datetime, timedelta, timezone +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.learning import validate_outcome +from devsquad.store import ConflictError, Store + + +NOW = datetime(2026, 9, 27, 16, 0, tzinfo=timezone.utc) + + +def final_outcome(): + return { + "schema_version": 1, + "outcome_id": "outcome-final", + "kind": "final", + "verdict": "succeeded", + "selection_mode": "automatic", + "observed_at": NOW.isoformat(), + "corrects_outcome_id": None, + "summary": "A repair attempt produced the accepted result.", + "criteria": [{ + "criterion_id": "checks", + "status": "passed", + "evidence_refs": ["receipt.json#criteria/checks"], + }], + "contributions": [ + { + "attempt_id": "attempt-original", + "role": "implementer", + "result": "failed", + "independent_success": False, + "evidence_refs": ["attempt-original.json"], + }, + { + "attempt_id": "attempt-repair", + "role": "implementer", + "result": "repair", + "independent_success": False, + "evidence_refs": ["attempt-repair.json"], + }, + ], + "lead_repairs": [], + "evidence_refs": ["receipt.json"], + } + + +class LearningContractTest(unittest.TestCase): + def test_outcome_contract_rejects_false_success_and_mutation(self): + normalized = validate_outcome(final_outcome(), now=NOW) + self.assertEqual(normalized["verdict"], "succeeded") + changed = final_outcome() + changed["criteria"][0]["status"] = "failed" + with self.assertRaisesRegex(ContractError, "must all pass"): + validate_outcome(changed, now=NOW) + changed = final_outcome() + changed["contributions"][0]["independent_success"] = True + with self.assertRaisesRegex(ContractError, "only a successful"): + validate_outcome(changed, now=NOW) + changed = final_outcome() + changed["extra"] = True + with self.assertRaisesRegex(ContractError, "fields invalid"): + validate_outcome(changed, now=NOW) + changed = final_outcome() + changed["observed_at"] = (NOW + timedelta(minutes=6)).isoformat() + with self.assertRaisesRegex(ContractError, "clock skew"): + validate_outcome(changed, now=NOW) + + def test_final_and_late_outcomes_are_append_only_and_attempt_bound(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + repository = path / "repo" + subprocess.run(["git", "init", "-q", str(repository)], check=True) + subprocess.run( + ["git", "-C", str(repository), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(repository), "config", "user.name", "Test"], + check=True, + ) + (repository / "README").write_text("fixture\n") + subprocess.run(["git", "-C", str(repository), "add", "README"], check=True) + subprocess.run(["git", "-C", str(repository), "commit", "-qm", "base"], check=True) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + claim = store.claim_start(repository, "outcome-run", {}, "owner") + run = store.run(claim.run_id) + store.connection.execute( + "UPDATE runs SET state='succeeded',phase=NULL WHERE id=?", + (claim.run_id,), + ) + for attempt_id, metadata in ( + ("attempt-original", {"failure": {"error": "seeded"}}), + ("attempt-repair", {}), + ): + store.connection.execute( + "INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token," + "status,heartbeat_at,package_digest,output_metadata,created_at,finished_at,role) " + "VALUES(?,?,?,?,?,'finished',?,?,?,?,?,'implementer')", + ( + attempt_id, claim.run_id, run["project_id"], str(repository), + f"token-{attempt_id}", NOW.isoformat(), "package", + json.dumps(metadata), NOW.isoformat(), NOW.isoformat(), + ), + ) + + false_credit = final_outcome() + false_credit["contributions"][0]["result"] = "successful" + false_credit["contributions"][0]["independent_success"] = True + with self.assertRaisesRegex(ConflictError, "failed attempt"): + store.record_outcome(claim.run_id, false_credit, now=NOW) + + first = store.record_outcome(claim.run_id, final_outcome(), now=NOW) + replay = store.record_outcome(claim.run_id, final_outcome(), now=NOW) + self.assertFalse(first["replayed"]) + self.assertTrue(replay["replayed"]) + changed = copy.deepcopy(final_outcome()) + changed["summary"] = "Changed after persistence." + with self.assertRaisesRegex(ConflictError, "different evidence"): + store.record_outcome(claim.run_id, changed, now=NOW) + + late = { + "schema_version": 1, + "outcome_id": "outcome-late-defect", + "kind": "late_correction", + "verdict": "escaped_defect", + "selection_mode": "automatic", + "observed_at": (NOW + timedelta(minutes=1)).isoformat(), + "corrects_outcome_id": "outcome-final", + "summary": "A defect escaped the original acceptance evidence.", + "criteria": [{ + "criterion_id": "checks", "status": "failed", + "evidence_refs": ["late-defect.json"], + }], + "contributions": [], + "lead_repairs": [], + "evidence_refs": ["late-defect.json"], + } + predating = copy.deepcopy(late) + predating["observed_at"] = (NOW - timedelta(seconds=1)).isoformat() + with self.assertRaisesRegex(ConflictError, "predates"): + store.record_outcome(claim.run_id, predating, now=NOW) + store.record_outcome( + claim.run_id, late, now=NOW + timedelta(minutes=1), + ) + history = store.outcomes_for_run(claim.run_id) + self.assertEqual( + [item["outcome"]["verdict"] for item in history], + ["succeeded", "escaped_defect"], + ) + self.assertFalse( + history[0]["outcome"]["contributions"][0]["independent_success"], + ) + + def test_migration_ten_creates_outcome_ledger(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + self.assertEqual( + store.connection.execute( + "SELECT MAX(version) FROM schema_migrations", + ).fetchone()[0], + 10, + ) + columns = { + row[1] for row in store.connection.execute("PRAGMA table_info(outcomes)") + } + self.assertTrue({"outcome_id", "payload_sha256", "corrects_outcome_id"} <= columns) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_store.py b/test/core/test_store.py index 94148a5..403ee59 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -482,8 +482,8 @@ def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): ) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 9) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(10,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 10) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(11,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -498,7 +498,7 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 9) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 10) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") @@ -515,7 +515,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],9) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],10) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From 45ebc9e2309598b5f45a92a917361df294adf7ab Mon Sep 17 00:00:00 2001 From: Dikshant Date: Sun, 27 Sep 2026 08:58:38 -0700 Subject: [PATCH 100/197] feat: report comparable learning outcomes --- plugin/core/src/devsquad/cli.py | 22 +++++ plugin/core/src/devsquad/learning.py | 131 +++++++++++++++++++++++++++ plugin/core/src/devsquad/service.py | 16 ++++ plugin/core/src/devsquad/store.py | 51 +++++++++++ test/core/test_cli.py | 26 ++++++ test/core/test_learning.py | 35 +++++-- test/core/test_service.py | 27 ++++++ 7 files changed, 302 insertions(+), 6 deletions(-) diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index ba47ce9..597dd4e 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -164,6 +164,15 @@ def command_capacity_observe(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).capacity_observe(observation)), 0 +def command_outcome_add(args: argparse.Namespace) -> tuple[dict, int]: + outcome = _read_json(args.file, "outcome file") + return envelope(data=_service(args).outcome_add(args.run, outcome)), 0 + + +def command_report(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).learning_report(args.project)), 0 + + def command_status(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).status(args.run)), 0 def command_events(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).events(args.run, args.after, args.limit)), 0 def command_result(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).result(args.run)), 0 @@ -260,6 +269,19 @@ def parser() -> argparse.ArgumentParser: observe.add_argument("--json", action="store_true") observe.add_argument("--runtime-dir", default=runtime_default) observe.set_defaults(func=command_capacity_observe) + outcome = sub.add_parser("outcome") + outcome_sub = outcome.add_subparsers(dest="outcome_command", required=True) + outcome_add = outcome_sub.add_parser("add") + outcome_add.add_argument("run") + outcome_add.add_argument("--file", required=True) + outcome_add.add_argument("--json", action="store_true") + outcome_add.add_argument("--runtime-dir", default=runtime_default) + outcome_add.set_defaults(func=command_outcome_add) + report = sub.add_parser("report") + report.add_argument("--project", required=True) + report.add_argument("--json", action="store_true") + report.add_argument("--runtime-dir", default=runtime_default) + report.set_defaults(func=command_report) handoff = sub.add_parser("handoff") handoff_sub = handoff.add_subparsers(dest="handoff_command", required=True) claim = handoff_sub.add_parser("claim") diff --git a/plugin/core/src/devsquad/learning.py b/plugin/core/src/devsquad/learning.py index 9f67a19..f3459c1 100644 --- a/plugin/core/src/devsquad/learning.py +++ b/plugin/core/src/devsquad/learning.py @@ -191,3 +191,134 @@ def validate_outcome( ), } return json.loads(canonical_json(normalized)) + + +def build_comparison_report( + *, + project_id: str | None, + project_path: str, + terminal_runs: list[dict[str, Any]], + outcome_records: list[dict[str, Any]], + attempt_profiles: dict[str, str | None], + generated_at: str, +) -> dict[str, Any]: + """Aggregate outcomes without conflating final success and worker quality.""" + if project_id is not None and (not isinstance(project_id, str) or not project_id): + raise ContractError("learning report project_id is invalid") + if not isinstance(project_path, str) or not project_path: + raise ContractError("learning report project_path is invalid") + _timestamp(generated_at, "generated_at") + if not isinstance(terminal_runs, list) or not isinstance(outcome_records, list): + raise ContractError("learning report inputs are invalid") + if not isinstance(attempt_profiles, dict): + raise ContractError("learning report attempt profiles are invalid") + + terminal_by_id = {} + for run in terminal_runs: + if (not isinstance(run, dict) or set(run) != {"run_id", "state"} + or not isinstance(run["run_id"], str) + or run["state"] not in FINAL_VERDICTS): + raise ContractError("learning report terminal run is invalid") + terminal_by_id[run["run_id"]] = run["state"] + + finals: dict[str, dict[str, Any]] = {} + corrections: dict[str, list[dict[str, Any]]] = {} + for record in outcome_records: + if (not isinstance(record, dict) or set(record) != {"run_id", "outcome"} + or record["run_id"] not in terminal_by_id + or not isinstance(record["outcome"], dict)): + raise ContractError("learning report outcome record is invalid") + outcome = record["outcome"] + if outcome.get("kind") == "final": + if record["run_id"] in finals: + raise ContractError("learning report has duplicate final outcomes") + finals[record["run_id"]] = outcome + elif outcome.get("kind") == "late_correction": + corrections.setdefault(record["run_id"], []).append(outcome) + else: + raise ContractError("learning report outcome kind is invalid") + + modes = { + mode: { + "sample_size": 0, + "succeeded": 0, + "failed": 0, + "cancelled": 0, + "escaped_defects": 0, + "success_rate": None, + } + for mode in sorted(SELECTION_MODES) + } + profiles: dict[str, dict[str, Any]] = {} + lead_repairs = 0 + missing_profile_contributions = 0 + finals_without_contributions = 0 + for run_id, final in finals.items(): + mode = final["selection_mode"] + mode_row = modes[mode] + mode_row["sample_size"] += 1 + mode_row[final["verdict"]] += 1 + escaped = sum( + correction["verdict"] == "escaped_defect" + for correction in corrections.get(run_id, []) + ) + mode_row["escaped_defects"] += escaped + lead_repairs += len(final["lead_repairs"]) + if not final["contributions"]: + finals_without_contributions += 1 + for contribution in final["contributions"]: + profile_id = attempt_profiles.get(contribution["attempt_id"]) + if profile_id is None: + missing_profile_contributions += 1 + continue + row = profiles.setdefault(profile_id, { + "contributions": 0, + "independent_successes": 0, + "failed": 0, + "repairs": 0, + "findings": 0, + }) + row["contributions"] += 1 + row["independent_successes"] += int( + contribution["independent_success"], + ) + if contribution["result"] == "failed": + row["failed"] += 1 + elif contribution["result"] == "repair": + row["repairs"] += 1 + elif contribution["result"] == "finding": + row["findings"] += 1 + for row in modes.values(): + if row["sample_size"]: + row["success_rate"] = row["succeeded"] / row["sample_size"] + + missing_run_ids = sorted(set(terminal_by_id) - set(finals)) + return { + "schema_version": 1, + "project_id": project_id, + "project_path": project_path, + "generated_at": generated_at, + "sample_size": len(finals), + "terminal_run_count": len(terminal_by_id), + "final_successes": sum( + outcome["verdict"] == "succeeded" for outcome in finals.values() + ), + "escaped_defects": sum( + correction["verdict"] == "escaped_defect" + for history in corrections.values() + for correction in history + ), + "lead_repairs": lead_repairs, + "selection_modes": modes, + "profiles": {profile_id: profiles[profile_id] for profile_id in sorted(profiles)}, + "missingness": { + "terminal_runs_without_final_outcome": len(missing_run_ids), + "terminal_run_ids_without_final_outcome": missing_run_ids, + "finals_without_contributions": finals_without_contributions, + "contributions_without_profile_id": missing_profile_contributions, + }, + "interpretation": { + "final_task_success_is_not_profile_success": True, + "selection_modes_are_not_pooled": True, + }, + } diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index e818230..77e7e4a 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -790,6 +790,22 @@ def capacity_observe(self, observation: dict[str, Any]) -> dict[str, Any]: finally: store.close() + def outcome_add(self, run_id: str, outcome: dict[str, Any]) -> dict[str, Any]: + store = self._store() + try: + return store.record_outcome(run_id, outcome) + finally: + store.close() + + def learning_report(self, project: str | Path) -> dict[str, Any]: + if not isinstance(project, (str, Path)): + raise ContractError("report project path is invalid") + store = self._store() + try: + return store.learning_report(Path(project)) + finally: + store.close() + @staticmethod def _status_capacity(store: Store, run: dict[str, Any]) -> dict[str, Any] | None: try: diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index ad09cc7..34c93ac 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1131,6 +1131,57 @@ def outcomes_for_run(self, run_id: str) -> list[dict[str, Any]]: ) ] + def learning_report( + self, project: Path, *, now: datetime | None = None, + ) -> dict[str, Any]: + """Build a read-only project comparison with explicit missingness.""" + from .learning import build_comparison_report + + common_dir = git_common_dir(project) + project_row = self.connection.execute( + "SELECT id FROM projects WHERE git_common_dir=?", (str(common_dir),), + ).fetchone() + if project_row is None: + terminal_runs = [] + outcome_records = [] + attempt_profiles = {} + project_id = None + else: + project_id = project_row["id"] + terminal_runs = [ + {"run_id": row["id"], "state": row["state"]} + for row in self.connection.execute( + "SELECT id,state FROM runs WHERE project_id=? " + "AND state IN ('succeeded','failed','cancelled') ORDER BY created_at,id", + (project_id,), + ) + ] + outcome_records = [ + {"run_id": row["run_id"], "outcome": json.loads(row["payload_json"])} + for row in self.connection.execute( + "SELECT o.run_id,o.payload_json FROM outcomes o " + "JOIN runs r ON r.id=o.run_id WHERE r.project_id=? " + "ORDER BY o.observed_at,o.id", + (project_id,), + ) + ] + attempt_profiles = { + row["id"]: row["profile_id"] + for row in self.connection.execute( + "SELECT a.id,a.profile_id FROM attempts a " + "JOIN runs r ON r.id=a.run_id WHERE r.project_id=?", + (project_id,), + ) + } + return build_comparison_report( + project_id=project_id, + project_path=str(project.resolve(strict=True)), + terminal_runs=terminal_runs, + outcome_records=outcome_records, + attempt_profiles=attempt_profiles, + generated_at=_authoritative_now(now).isoformat(), + ) + def worker_invocations(self, run_id: str) -> int: """Count attempts whose durable runner actually crossed the launch fence.""" return self.connection.execute( diff --git a/test/core/test_cli.py b/test/core/test_cli.py index d1c2293..ad9b8e0 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -125,6 +125,32 @@ def test_capacity_observe_dispatches_file(self): self.assert_success_envelope(payload, response) service.capacity_observe.assert_called_once_with(observation) + def test_outcome_add_and_report_dispatch(self): + outcome = {"schema_version": 1, "outcome_id": "outcome-1"} + outcome_file = self.root / "outcome.json" + outcome_file.write_text(json.dumps(outcome)) + service = mock.Mock() + service.outcome_add.return_value = {"run_id": "run-1", "replayed": False} + code, payload, stderr = self.invoke([ + "outcome", "add", "run-1", "--file", str(outcome_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope( + payload, {"run_id": "run-1", "replayed": False}, + ) + service.outcome_add.assert_called_once_with("run-1", outcome) + + service = mock.Mock() + service.learning_report.return_value = {"sample_size": 3} + code, payload, stderr = self.invoke([ + "report", "--project", str(self.root), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, {"sample_size": 3}) + service.learning_report.assert_called_once_with(str(self.root)) + def test_handoff_claim_renew_and_complete_dispatch_parsed_objects(self): claim_payload = { "schema_version": 1, diff --git a/test/core/test_learning.py b/test/core/test_learning.py index 62ad73e..3c75b4f 100644 --- a/test/core/test_learning.py +++ b/test/core/test_learning.py @@ -99,18 +99,18 @@ def test_final_and_late_outcomes_are_append_only_and_attempt_bound(self): "UPDATE runs SET state='succeeded',phase=NULL WHERE id=?", (claim.run_id,), ) - for attempt_id, metadata in ( - ("attempt-original", {"failure": {"error": "seeded"}}), - ("attempt-repair", {}), + for attempt_id, profile_id, metadata in ( + ("attempt-original", "profile-original", {"failure": {"error": "seeded"}}), + ("attempt-repair", "profile-repair", {}), ): store.connection.execute( "INSERT INTO attempts(id,run_id,project_id,worktree_path,attempt_token," - "status,heartbeat_at,package_digest,output_metadata,created_at,finished_at,role) " - "VALUES(?,?,?,?,?,'finished',?,?,?,?,?,'implementer')", + "status,heartbeat_at,package_digest,output_metadata,created_at,finished_at," + "role,profile_id) VALUES(?,?,?,?,?,'finished',?,?,?,?,?,'implementer',?)", ( attempt_id, claim.run_id, run["project_id"], str(repository), f"token-{attempt_id}", NOW.isoformat(), "package", - json.dumps(metadata), NOW.isoformat(), NOW.isoformat(), + json.dumps(metadata), NOW.isoformat(), NOW.isoformat(), profile_id, ), ) @@ -161,6 +161,29 @@ def test_final_and_late_outcomes_are_append_only_and_attempt_bound(self): self.assertFalse( history[0]["outcome"]["contributions"][0]["independent_success"], ) + missing = store.claim_start(repository, "missing-outcome", {}, "owner") + store.connection.execute( + "UPDATE runs SET state='failed',phase=NULL WHERE id=?", + (missing.run_id,), + ) + report = store.learning_report( + repository, now=NOW + timedelta(minutes=2), + ) + self.assertEqual(report["sample_size"], 1) + self.assertEqual(report["terminal_run_count"], 2) + self.assertEqual(report["final_successes"], 1) + self.assertEqual(report["escaped_defects"], 1) + self.assertEqual( + report["selection_modes"]["automatic"]["success_rate"], 1.0, + ) + self.assertEqual(report["profiles"]["profile-original"]["failed"], 1) + self.assertEqual(report["profiles"]["profile-repair"]["repairs"], 1) + self.assertEqual( + report["missingness"]["terminal_runs_without_final_outcome"], 1, + ) + self.assertTrue( + report["interpretation"]["final_task_success_is_not_profile_success"], + ) def test_migration_ten_creates_outcome_ledger(self): with tempfile.TemporaryDirectory() as root: diff --git a/test/core/test_service.py b/test/core/test_service.py index baace58..6199e8f 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -165,6 +165,33 @@ def test_capacity_observation_drives_preflight_and_status_evidence(self): self.assertEqual(current["status"], "available") self.assertEqual(current["windows"][0]["window_id"], "short") + def test_outcome_add_and_project_report_share_saved_ledger(self): + started = self.service.start( + self.task, "learning-report", _internal_fake_delay=.01, + ) + self.wait_state(started["run_id"], {"succeeded"}) + now = datetime.now(timezone.utc) + recorded = self.service.outcome_add(started["run_id"], { + "schema_version": 1, + "outcome_id": "service-final-outcome", + "kind": "final", + "verdict": "succeeded", + "selection_mode": "automatic", + "observed_at": now.isoformat(), + "corrects_outcome_id": None, + "summary": "The saved fixture run completed successfully.", + "criteria": [], + "contributions": [], + "lead_repairs": [], + "evidence_refs": ["result-receipt.json"], + }) + self.assertFalse(recorded["replayed"]) + report = self.service.learning_report(self.repo) + self.assertEqual(report["sample_size"], 1) + self.assertEqual(report["terminal_run_count"], 1) + self.assertEqual(report["final_successes"], 1) + self.assertEqual(report["missingness"]["finals_without_contributions"], 1) + def test_abandoned_preparation_is_reclaimed_from_the_submitted_request(self): store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") try: From edfb3f37641d9e27f59088c27b06009666589fa1 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 01:42:31 -0700 Subject: [PATCH 101/197] feat: evaluate bounded learning experiments --- plugin/core/src/devsquad/cli.py | 12 + plugin/core/src/devsquad/learning.py | 241 ++++++++++++++++++ .../devsquad/migrations/011_experiments.sql | 17 ++ plugin/core/src/devsquad/service.py | 8 + plugin/core/src/devsquad/store.py | 89 ++++++- test/core/test_capacity.py | 2 +- test/core/test_cli.py | 29 ++- test/core/test_handoff_store.py | 6 +- test/core/test_learning.py | 175 ++++++++++++- test/core/test_store.py | 8 +- 10 files changed, 573 insertions(+), 14 deletions(-) create mode 100644 plugin/core/src/devsquad/migrations/011_experiments.sql diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 597dd4e..ce3061a 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -173,6 +173,11 @@ def command_report(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).learning_report(args.project)), 0 +def command_policy_evaluate(args: argparse.Namespace) -> tuple[dict, int]: + experiment = _read_json(args.experiment, "experiment file") + return envelope(data=_service(args).policy_evaluate(experiment)), 0 + + def command_status(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).status(args.run)), 0 def command_events(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).events(args.run, args.after, args.limit)), 0 def command_result(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).result(args.run)), 0 @@ -282,6 +287,13 @@ def parser() -> argparse.ArgumentParser: report.add_argument("--json", action="store_true") report.add_argument("--runtime-dir", default=runtime_default) report.set_defaults(func=command_report) + policy = sub.add_parser("policy") + policy_sub = policy.add_subparsers(dest="policy_command", required=True) + evaluate = policy_sub.add_parser("evaluate") + evaluate.add_argument("--experiment", required=True) + evaluate.add_argument("--json", action="store_true") + evaluate.add_argument("--runtime-dir", default=runtime_default) + evaluate.set_defaults(func=command_policy_evaluate) handoff = sub.add_parser("handoff") handoff_sub = handoff.add_subparsers(dest="handoff_command", required=True) claim = handoff_sub.add_parser("claim") diff --git a/plugin/core/src/devsquad/learning.py b/plugin/core/src/devsquad/learning.py index f3459c1..c9aa76a 100644 --- a/plugin/core/src/devsquad/learning.py +++ b/plugin/core/src/devsquad/learning.py @@ -3,7 +3,9 @@ from __future__ import annotations from datetime import datetime, timedelta, timezone +import hashlib import json +from pathlib import Path from typing import Any from .contracts import ContractError @@ -22,6 +24,11 @@ CONTRIBUTION_RESULTS = {"failed", "successful", "repair", "finding", "neutral"} ROLES = {"worker", "implementer", "reviewer", "lead", "researcher"} MAX_CLOCK_SKEW = timedelta(minutes=5) +EXPERIMENT_FIELDS = { + "schema_version", "experiment_id", "project_path", "question", "hypothesis", + "evidence_availability", "variable", "cases", "gate", "budget", + "rollback_target", +} def _now(value: datetime | None) -> datetime: @@ -322,3 +329,237 @@ def build_comparison_report( "selection_modes_are_not_pooled": True, }, } + + +def validate_experiment(value: dict[str, Any]) -> dict[str, Any]: + """Validate a predeclared one-variable, paired outcome experiment.""" + if not isinstance(value, dict) or set(value) != EXPERIMENT_FIELDS: + raise ContractError("experiment fields are invalid") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("experiment schema_version is invalid") + experiment_id = _identifier(value["experiment_id"], "experiment_id") + project_path = value["project_path"] + if (not isinstance(project_path, str) or not project_path + or not Path(project_path).is_absolute()): + raise ContractError("experiment project_path must be absolute") + for field in ("question", "hypothesis"): + if not isinstance(value[field], str) or not value[field].strip(): + raise ContractError(f"experiment {field} must be a non-empty string") + if (not isinstance(value["evidence_availability"], str) + or value["evidence_availability"] not in { + "local", "tracked_fixture", "unavailable", + }): + raise ContractError("experiment evidence_availability is invalid") + variable = value["variable"] + if not isinstance(variable, dict) or set(variable) != { + "kind", "alias", "control_profile_id", "candidate_profile_id", + }: + raise ContractError("experiment variable fields are invalid") + if variable["kind"] != "profile_binding": + raise ContractError("experiment variable kind is invalid") + for field in ("alias", "control_profile_id", "candidate_profile_id"): + _identifier(variable[field], f"variable.{field}") + if variable["control_profile_id"] == variable["candidate_profile_id"]: + raise ContractError("experiment control and candidate must differ") + + budget = value["budget"] + if not isinstance(budget, dict) or set(budget) != { + "max_cases", "max_worker_invocations", "wall_seconds", + }: + raise ContractError("experiment budget fields are invalid") + for field in ("max_cases", "wall_seconds"): + if type(budget[field]) is not int or budget[field] < 1: + raise ContractError(f"experiment budget {field} must be positive") + if type(budget["max_worker_invocations"]) is not int or budget["max_worker_invocations"] < 0: + raise ContractError("experiment max_worker_invocations must be non-negative") + + cases = value["cases"] + if not isinstance(cases, list) or not cases or len(cases) > budget["max_cases"]: + raise ContractError("experiment cases exceed the bounded case budget") + case_ids = set() + normalized_cases = [] + split_counts = {"evaluation": 0, "held_out": 0} + for case in cases: + if not isinstance(case, dict) or set(case) != { + "case_id", "split", "control_outcome_id", "candidate_outcome_id", + }: + raise ContractError("experiment case fields are invalid") + case_id = _identifier(case["case_id"], "case_id") + if case_id in case_ids: + raise ContractError("experiment case ids must be unique") + case_ids.add(case_id) + if not isinstance(case["split"], str) or case["split"] not in split_counts: + raise ContractError("experiment case split is invalid") + split_counts[case["split"]] += 1 + control_id = _identifier(case["control_outcome_id"], "control_outcome_id") + candidate_id = _identifier( + case["candidate_outcome_id"], "candidate_outcome_id", + ) + if control_id == candidate_id: + raise ContractError("experiment paired outcomes must differ") + normalized_cases.append({ + "case_id": case_id, + "split": case["split"], + "control_outcome_id": control_id, + "candidate_outcome_id": candidate_id, + }) + + gate = value["gate"] + if not isinstance(gate, dict) or set(gate) != { + "min_evaluation_pairs", "min_held_out_pairs", "noninferiority_margin", + "minimum_success_gain", "max_candidate_escaped_defects", + }: + raise ContractError("experiment gate fields are invalid") + for field, split in ( + ("min_evaluation_pairs", "evaluation"), + ("min_held_out_pairs", "held_out"), + ): + if (type(gate[field]) is not int or gate[field] < 1 + or gate[field] > split_counts[split]): + raise ContractError(f"experiment gate {field} is invalid") + for field in ("noninferiority_margin", "minimum_success_gain"): + number = gate[field] + if (isinstance(number, bool) or not isinstance(number, (int, float)) + or not 0 <= number <= 1): + raise ContractError(f"experiment gate {field} must be between zero and one") + if (type(gate["max_candidate_escaped_defects"]) is not int + or gate["max_candidate_escaped_defects"] < 0): + raise ContractError("experiment escaped-defect gate is invalid") + + rollback = value["rollback_target"] + if not isinstance(rollback, dict) or set(rollback) != { + "profile_id", "binding_version", + }: + raise ContractError("experiment rollback target fields are invalid") + if rollback["profile_id"] != variable["control_profile_id"]: + raise ContractError("experiment rollback target must be the control profile") + if type(rollback["binding_version"]) is not int or rollback["binding_version"] < 1: + raise ContractError("experiment rollback binding_version is invalid") + return json.loads(canonical_json({ + **value, + "experiment_id": experiment_id, + "variable": dict(variable), + "cases": normalized_cases, + "gate": dict(gate), + "budget": dict(budget), + "rollback_target": dict(rollback), + })) + + +def evaluate_experiment( + experiment: dict[str, Any], + outcome_chains: dict[str, dict[str, Any]], + *, + evaluated_at: str, +) -> dict[str, Any]: + """Evaluate a frozen paired experiment without changing active policy.""" + spec = validate_experiment(experiment) + _timestamp(evaluated_at, "evaluated_at") + if not isinstance(outcome_chains, dict): + raise ContractError("experiment outcome chains are invalid") + rows = [] + metrics = { + split: { + "declared_pairs": 0, + "available_pairs": 0, + "control_successes": 0, + "candidate_successes": 0, + "candidate_escaped_defects": 0, + "control_success_rate": None, + "candidate_success_rate": None, + "success_gain": None, + } + for split in ("evaluation", "held_out") + } + failures = [] + for case in spec["cases"]: + split = case["split"] + metrics[split]["declared_pairs"] += 1 + control = outcome_chains.get(case["control_outcome_id"]) + candidate = outcome_chains.get(case["candidate_outcome_id"]) + missing = [] + if control is None: + missing.append("control") + if candidate is None: + missing.append("candidate") + row = {**case, "status": "missing" if missing else "available", "missing": missing} + if missing: + failures.append({"case_id": case["case_id"], "reason": "missing_outcome"}) + rows.append(row) + continue + for arm, chain in (("control", control), ("candidate", candidate)): + if (not isinstance(chain, dict) or set(chain) != {"final", "late_corrections"} + or not isinstance(chain["final"], dict) + or not isinstance(chain["late_corrections"], list)): + raise ContractError("experiment outcome chain is invalid") + if chain["final"].get("kind") != "final": + raise ContractError("experiment arm must reference a final outcome") + if chain["final"].get("selection_mode") != "experimental": + raise ContractError("experiment outcomes must be explicitly experimental") + if any(not isinstance(correction, dict) + for correction in chain["late_corrections"]): + raise ContractError("experiment late corrections are invalid") + row[f"{arm}_verdict"] = chain["final"]["verdict"] + row[f"{arm}_escaped_defects"] = sum( + correction.get("verdict") == "escaped_defect" + for correction in chain["late_corrections"] + ) + metrics[split]["available_pairs"] += 1 + metrics[split]["control_successes"] += int(row["control_verdict"] == "succeeded") + metrics[split]["candidate_successes"] += int( + row["candidate_verdict"] == "succeeded", + ) + metrics[split]["candidate_escaped_defects"] += row[ + "candidate_escaped_defects" + ] + if row["candidate_verdict"] != "succeeded": + failures.append({ + "case_id": case["case_id"], "reason": "candidate_not_successful", + }) + if row["candidate_escaped_defects"]: + failures.append({ + "case_id": case["case_id"], "reason": "candidate_escaped_defect", + }) + rows.append(row) + + reasons = [] + total_candidate_escaped = 0 + for split, row in metrics.items(): + minimum = spec["gate"][ + "min_evaluation_pairs" if split == "evaluation" else "min_held_out_pairs" + ] + if row["available_pairs"] < minimum: + reasons.append(f"insufficient_{split}_pairs") + continue + row["control_success_rate"] = row["control_successes"] / row["available_pairs"] + row["candidate_success_rate"] = ( + row["candidate_successes"] / row["available_pairs"] + ) + row["success_gain"] = row["candidate_success_rate"] - row["control_success_rate"] + if (row["candidate_success_rate"] + spec["gate"]["noninferiority_margin"] + < row["control_success_rate"]): + reasons.append(f"{split}_noninferiority_failed") + if row["success_gain"] < spec["gate"]["minimum_success_gain"]: + reasons.append(f"{split}_minimum_gain_failed") + total_candidate_escaped += row["candidate_escaped_defects"] + if total_candidate_escaped > spec["gate"]["max_candidate_escaped_defects"]: + reasons.append("candidate_escaped_defect_limit_exceeded") + + verdict = "promotion_proposal" if not reasons else "no_change" + return { + "schema_version": 1, + "experiment_id": spec["experiment_id"], + "spec_sha256": hashlib.sha256( + canonical_json(spec).encode(), + ).hexdigest(), + "evaluated_at": evaluated_at, + "verdict": verdict, + "active_policy_changed": False, + "reasons": sorted(set(reasons)), + "metrics": metrics, + "cases": rows, + "failures": failures, + "variable": spec["variable"], + "rollback_target": spec["rollback_target"], + "evidence_availability": spec["evidence_availability"], + } diff --git a/plugin/core/src/devsquad/migrations/011_experiments.sql b/plugin/core/src/devsquad/migrations/011_experiments.sql new file mode 100644 index 0000000..0c78530 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/011_experiments.sql @@ -0,0 +1,17 @@ +CREATE TABLE experiments ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + experiment_id TEXT NOT NULL UNIQUE, + project_id TEXT REFERENCES projects(id), + project_path TEXT NOT NULL, + spec_json TEXT NOT NULL, + spec_sha256 TEXT NOT NULL + CHECK(length(spec_sha256) = 64 AND spec_sha256 NOT GLOB '*[^0-9a-f]*'), + evaluation_json TEXT NOT NULL, + evaluation_sha256 TEXT NOT NULL + CHECK(length(evaluation_sha256) = 64 AND evaluation_sha256 NOT GLOB '*[^0-9a-f]*'), + verdict TEXT NOT NULL CHECK(verdict IN ('no_change', 'promotion_proposal')), + recorded_at TEXT NOT NULL +); + +CREATE INDEX experiments_project_history +ON experiments(project_id, recorded_at, id); diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 77e7e4a..8a7b788 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -806,6 +806,14 @@ def learning_report(self, project: str | Path) -> dict[str, Any]: finally: store.close() + def policy_evaluate(self, experiment: dict[str, Any]) -> dict[str, Any]: + """Evaluate and save one frozen learning experiment without promotion.""" + store = self._store() + try: + return store.evaluate_learning_experiment(experiment) + finally: + store.close() + @staticmethod def _status_capacity(store: Store, run: dict[str, Any]) -> dict[str, Any] | None: try: diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 34c93ac..6822c11 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 10 +SUPPORTED_SCHEMA_VERSION = 11 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -1182,6 +1182,93 @@ def learning_report( generated_at=_authoritative_now(now).isoformat(), ) + def evaluate_learning_experiment( + self, experiment: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """Persist one deterministic, replay-safe experiment evaluation.""" + from .learning import evaluate_experiment, validate_experiment + + spec = validate_experiment(experiment) + spec_json = canonical_json(spec) + spec_sha256 = hashlib.sha256(spec_json.encode()).hexdigest() + current = _authoritative_now(now) + common_dir = git_common_dir(Path(spec["project_path"])) + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT spec_json,evaluation_json,evaluation_sha256,recorded_at " + "FROM experiments WHERE experiment_id=?", + (spec["experiment_id"],), + ).fetchone() + if existing is not None: + if existing["spec_json"] != spec_json: + raise ConflictError( + "experiment id was already used with a different specification", + ) + self.connection.execute("COMMIT") + return { + "experiment": spec, + "evaluation": json.loads(existing["evaluation_json"]), + "evaluation_sha256": existing["evaluation_sha256"], + "recorded_at": existing["recorded_at"], + "replayed": True, + } + project = self.connection.execute( + "SELECT id FROM projects WHERE git_common_dir=?", (str(common_dir),), + ).fetchone() + project_id = project["id"] if project is not None else None + records = [] + if project_id is not None: + records = self.connection.execute( + "SELECT o.payload_json FROM outcomes o " + "JOIN runs r ON r.id=o.run_id WHERE r.project_id=? " + "ORDER BY o.observed_at,o.id", + (project_id,), + ).fetchall() + chains: dict[str, dict[str, Any]] = {} + corrections: dict[str, list[dict[str, Any]]] = {} + for record in records: + outcome = json.loads(record["payload_json"]) + if outcome["kind"] == "final": + chains[outcome["outcome_id"]] = { + "final": outcome, + "late_corrections": [], + } + else: + corrections.setdefault( + outcome["corrects_outcome_id"], [], + ).append(outcome) + for outcome_id, history in corrections.items(): + if outcome_id in chains: + chains[outcome_id]["late_corrections"] = history + evaluation = evaluate_experiment( + spec, chains, evaluated_at=current.isoformat(), + ) + evaluation_json = canonical_json(evaluation) + evaluation_sha256 = hashlib.sha256(evaluation_json.encode()).hexdigest() + recorded_at = current.isoformat() + self.connection.execute( + "INSERT INTO experiments(experiment_id,project_id,project_path,spec_json," + "spec_sha256,evaluation_json,evaluation_sha256,verdict,recorded_at) " + "VALUES(?,?,?,?,?,?,?,?,?)", + ( + spec["experiment_id"], project_id, spec["project_path"], + spec_json, spec_sha256, evaluation_json, evaluation_sha256, + evaluation["verdict"], recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "experiment": spec, + "evaluation": evaluation, + "evaluation_sha256": evaluation_sha256, + "recorded_at": recorded_at, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + def worker_invocations(self, run_id: str) -> int: """Count attempts whose durable runner actually crossed the launch fence.""" return self.connection.execute( diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index 87fec4e..7de13b5 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -184,7 +184,7 @@ def test_migration_nine_creates_capacity_ledger(self): version = store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 10) + self.assertEqual(version, 11) tables = { row[0] for row in store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_cli.py b/test/core/test_cli.py index ad9b8e0..761295a 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -151,6 +151,31 @@ def test_outcome_add_and_report_dispatch(self): self.assert_success_envelope(payload, {"sample_size": 3}) service.learning_report.assert_called_once_with(str(self.root)) + def test_policy_evaluate_dispatches_frozen_experiment(self): + experiment = { + "schema_version": 1, + "experiment_id": "experiment-1", + "project_path": str(self.root), + } + experiment_file = self.root / "experiment.json" + experiment_file.write_text(json.dumps(experiment)) + service = mock.Mock() + response = { + "evaluation": { + "verdict": "no_change", + "active_policy_changed": False, + }, + "replayed": False, + } + service.policy_evaluate.return_value = response + code, payload, stderr = self.invoke([ + "policy", "evaluate", "--experiment", str(experiment_file), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.policy_evaluate.assert_called_once_with(experiment) + def test_handoff_claim_renew_and_complete_dispatch_parsed_objects(self): claim_payload = { "schema_version": 1, @@ -413,7 +438,7 @@ def build_python(): return candidate return None - def test_installed_wheel_contains_and_applies_migrations_through_ten(self): + def test_installed_wheel_contains_and_applies_migrations_through_eleven(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") @@ -455,7 +480,7 @@ def test_installed_wheel_contains_and_applies_migrations_through_ten(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 10 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 11 attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 244b0ae..14c7329 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 10, + 11, ) tables = { row[0] @@ -618,7 +618,7 @@ def build_python(): return candidate return None - def test_installed_wheel_applies_schema_four_to_ten(self): + def test_installed_wheel_applies_schema_four_to_eleven(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel") @@ -684,7 +684,7 @@ def test_installed_wheel_applies_schema_four_to_ten(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 10 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 11 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_learning.py b/test/core/test_learning.py index 3c75b4f..d07ac4b 100644 --- a/test/core/test_learning.py +++ b/test/core/test_learning.py @@ -11,7 +11,11 @@ sys.path.insert(0, str(ROOT / "plugin/core/src")) from devsquad.contracts import ContractError -from devsquad.learning import validate_outcome +from devsquad.learning import ( + evaluate_experiment, + validate_experiment, + validate_outcome, +) from devsquad.store import ConflictError, Store @@ -54,6 +58,66 @@ def final_outcome(): } +def experiment(project_path): + return { + "schema_version": 1, + "experiment_id": "experiment-profile-b", + "project_path": str(project_path), + "question": "Does profile B improve successful outcomes?", + "hypothesis": "Profile B is non-inferior and improves paired success.", + "evidence_availability": "tracked_fixture", + "variable": { + "kind": "profile_binding", + "alias": "review.deep", + "control_profile_id": "profile-a", + "candidate_profile_id": "profile-b", + }, + "cases": [ + { + "case_id": "eval-1", "split": "evaluation", + "control_outcome_id": "control-eval-1", + "candidate_outcome_id": "candidate-eval-1", + }, + { + "case_id": "eval-2", "split": "evaluation", + "control_outcome_id": "control-eval-2", + "candidate_outcome_id": "candidate-eval-2", + }, + { + "case_id": "hold-1", "split": "held_out", + "control_outcome_id": "control-hold-1", + "candidate_outcome_id": "candidate-hold-1", + }, + ], + "gate": { + "min_evaluation_pairs": 2, + "min_held_out_pairs": 1, + "noninferiority_margin": 0.0, + "minimum_success_gain": 0.5, + "max_candidate_escaped_defects": 0, + }, + "budget": { + "max_cases": 3, + "max_worker_invocations": 0, + "wall_seconds": 60, + }, + "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, + } + + +def experimental_final(outcome_id, verdict): + value = final_outcome() + value.update({ + "outcome_id": outcome_id, + "verdict": verdict, + "selection_mode": "experimental", + "summary": f"Experimental fixture {outcome_id} was {verdict}.", + "criteria": [], + "contributions": [], + }) + return value + + class LearningContractTest(unittest.TestCase): def test_outcome_contract_rejects_false_success_and_mutation(self): normalized = validate_outcome(final_outcome(), now=NOW) @@ -75,6 +139,104 @@ def test_outcome_contract_rejects_false_success_and_mutation(self): with self.assertRaisesRegex(ContractError, "clock skew"): validate_outcome(changed, now=NOW) + def test_experiment_gate_promotes_only_complete_held_out_evidence(self): + spec = experiment(Path("/tmp/experiment-project")) + validate_experiment(spec) + chains = {} + for case in spec["cases"]: + chains[case["control_outcome_id"]] = { + "final": experimental_final(case["control_outcome_id"], "failed"), + "late_corrections": [], + } + chains[case["candidate_outcome_id"]] = { + "final": experimental_final(case["candidate_outcome_id"], "succeeded"), + "late_corrections": [], + } + promoted = evaluate_experiment(spec, chains, evaluated_at=NOW.isoformat()) + self.assertEqual(promoted["verdict"], "promotion_proposal") + self.assertFalse(promoted["active_policy_changed"]) + self.assertEqual( + promoted["rollback_target"], + {"profile_id": "profile-a", "binding_version": 7}, + ) + + missing = dict(chains) + del missing["candidate-hold-1"] + no_change = evaluate_experiment(spec, missing, evaluated_at=NOW.isoformat()) + self.assertEqual(no_change["verdict"], "no_change") + self.assertIn("insufficient_held_out_pairs", no_change["reasons"]) + + escaped = copy.deepcopy(chains) + escaped["candidate-hold-1"]["late_corrections"] = [{ + "verdict": "escaped_defect", + }] + no_change = evaluate_experiment(spec, escaped, evaluated_at=NOW.isoformat()) + self.assertEqual(no_change["verdict"], "no_change") + self.assertIn( + "candidate_escaped_defect_limit_exceeded", no_change["reasons"], + ) + + invalid = experiment(Path("/tmp/experiment-project")) + invalid["rollback_target"]["profile_id"] = "profile-b" + with self.assertRaisesRegex(ContractError, "control profile"): + validate_experiment(invalid) + invalid = experiment(Path("/tmp/experiment-project")) + invalid["evidence_availability"] = [] + with self.assertRaisesRegex(ContractError, "evidence_availability"): + validate_experiment(invalid) + invalid = experiment(Path("/tmp/experiment-project")) + invalid["cases"][0]["split"] = [] + with self.assertRaisesRegex(ContractError, "case split"): + validate_experiment(invalid) + invalid_chains = copy.deepcopy(chains) + invalid_chains["candidate-hold-1"]["late_corrections"] = [None] + with self.assertRaisesRegex(ContractError, "late corrections"): + evaluate_experiment(spec, invalid_chains, evaluated_at=NOW.isoformat()) + + def test_experiment_evaluation_is_persisted_and_replay_safe(self): + with tempfile.TemporaryDirectory() as root: + path = Path(root) + repository = path / "repo" + subprocess.run(["git", "init", "-q", str(repository)], check=True) + subprocess.run( + ["git", "-C", str(repository), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(repository), "config", "user.name", "Test"], + check=True, + ) + (repository / "README").write_text("fixture\n") + subprocess.run(["git", "-C", str(repository), "add", "README"], check=True) + subprocess.run(["git", "-C", str(repository), "commit", "-qm", "base"], check=True) + store = Store(path / "state.sqlite3", path / "artifacts") + self.addCleanup(store.close) + spec = experiment(repository) + for case in spec["cases"]: + for arm, verdict in (("control", "failed"), ("candidate", "succeeded")): + outcome_id = case[f"{arm}_outcome_id"] + claim = store.claim_start( + repository, f"run-{outcome_id}", {}, "owner", + ) + store.connection.execute( + "UPDATE runs SET state=?,phase=NULL WHERE id=?", + (verdict, claim.run_id), + ) + store.record_outcome( + claim.run_id, + experimental_final(outcome_id, verdict), + now=NOW, + ) + first = store.evaluate_learning_experiment(spec, now=NOW) + replay = store.evaluate_learning_experiment(spec, now=NOW) + self.assertEqual(first["evaluation"]["verdict"], "promotion_proposal") + self.assertFalse(first["replayed"]) + self.assertTrue(replay["replayed"]) + changed = copy.deepcopy(spec) + changed["hypothesis"] = "Mutated after evaluation." + with self.assertRaisesRegex(ConflictError, "different specification"): + store.evaluate_learning_experiment(changed, now=NOW) + def test_final_and_late_outcomes_are_append_only_and_attempt_bound(self): with tempfile.TemporaryDirectory() as root: path = Path(root) @@ -185,7 +347,7 @@ def test_final_and_late_outcomes_are_append_only_and_attempt_bound(self): report["interpretation"]["final_task_success_is_not_profile_success"], ) - def test_migration_ten_creates_outcome_ledger(self): + def test_current_schema_contains_outcome_ledger(self): with tempfile.TemporaryDirectory() as root: path = Path(root) store = Store(path / "state.sqlite3", path / "artifacts") @@ -194,12 +356,19 @@ def test_migration_ten_creates_outcome_ledger(self): store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0], - 10, + 11, ) columns = { row[1] for row in store.connection.execute("PRAGMA table_info(outcomes)") } self.assertTrue({"outcome_id", "payload_sha256", "corrects_outcome_id"} <= columns) + experiment_columns = { + row[1] + for row in store.connection.execute("PRAGMA table_info(experiments)") + } + self.assertTrue({ + "experiment_id", "spec_sha256", "evaluation_sha256", "verdict", + } <= experiment_columns) if __name__ == "__main__": diff --git a/test/core/test_store.py b/test/core/test_store.py index 403ee59..efb2335 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -482,8 +482,8 @@ def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): ) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 10) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(11,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 11) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(12,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -498,7 +498,7 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 10) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 11) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") @@ -515,7 +515,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],10) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],11) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From f01e39abe16bb5f95c085016cf3e89afd26a9216 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 01:43:45 -0700 Subject: [PATCH 102/197] docs: checkpoint M6 experiment evaluation --- docs/plans/engineering-team/M6-STATUS.md | 43 ++++++++++++++++++------ docs/plans/engineering-team/RESUME.md | 22 ++++++++---- docs/plans/engineering-team/backlog.json | 9 +++++ 3 files changed, 58 insertions(+), 16 deletions(-) diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md index fe4eccd..eca76e6 100644 --- a/docs/plans/engineering-team/M6-STATUS.md +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -1,7 +1,8 @@ # M6 implementation status -M6 is **in progress**. Shared capacity observations, reservations and routing -are verified; outcomes, experiments, lifecycle promotion/rollback and the +M6 is **in progress**. Shared capacity, the append-only outcome ledger, +comparison reports and replay-safe one-variable experiment evaluation are +verified. Draft proposal generation, lifecycle promotion/rollback and the default-off decision helper remain open. | Requirement | Planned evidence | Status | @@ -10,8 +11,10 @@ default-off decision helper remain open. | Shared transactional reservations | Two projects share one pool; one-slot races produce one owner; schema-8 active attempts survive migration; ambiguous ownership retains the reservation | verified at `26ce5cf` | | Capacity-aware routing | All applicable windows and profile sublimits affect deterministic selection; a later exhausted observation blocks reservation; paid API remains policy-gated | verified at `d170c01` | | Public observation/status surface | `squad capacity observe --file FILE`; replay-safe persistence; status shows frozen and current detailed evidence | verified at `d170c01` | -| Final and late outcomes | Preserve attempt contribution, lead repair, final success and escaped-defect corrections without crediting failed attempts | pending | -| Comparison reports and proposals | Sample sizes, missingness, selection mode, one-variable experiment, held-out rerun, no-change/promotion proposal and rollback target | pending | +| Final and late outcomes | Preserve attempt contribution, lead repair, final success and escaped-defect corrections without crediting failed attempts | verified at `d621df2` | +| Comparison reports | Sample sizes, missingness and separated automatic/pinned/experimental evidence | verified at `45ebc9e` | +| Frozen experiment evaluation | One-variable paired evaluation/held-out cases, failure evidence, no-change or promotion-proposal verdict and rollback target; evaluation never changes active policy | verified at `edfb3f3` | +| Draft proposals and held-out rerun | `learn propose`, review packet and post-change held-out/rollback evidence | pending | | Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | pending | | Decision helper M6-D1 | Default-off typed contract, fake adapter, cache/accounting and authority/integrity tests | in progress; synthetic Jev probe mechanics only | | Jev M6-D2 | One capped synthetic request with exact model/usage/latency/cost receipt | blocked on `TYPESAFE_API_KEY` | @@ -41,11 +44,31 @@ replay-safe JSON, and run status returns both frozen and current windows. The checkpoint gate is **255 core tests with 2 optional-SDK skips** and ResourceWarning promoted to error, plus **220/220 Bash assertions**. +## Outcome and experiment checkpoint + +Schema 10 adds append-only final and late-correction outcomes bound to saved +runs and attempts. A repaired task can succeed without falsely crediting the +failed original attempt, while a later escaped defect remains attached to the +original final verdict. `squad report --project PATH` reports sample size, +missingness and explicit contribution credit separately for automatic, +pinned and experimental selections. + +Schema 11 adds immutable experiment specifications and replay-safe evaluation. +`squad policy evaluate --experiment FILE` compares declared control/candidate +pairs across evaluation and held-out splits, records missing and failed cases, +enforces non-inferiority/gain/escaped-defect gates and preserves an explicit +rollback version. Its output is only `no_change` or `promotion_proposal` and +always records `active_policy_changed: false`; reusing an experiment ID with a +different specification conflicts. + +The combined checkpoint gate is **263 core tests with 2 optional-SDK skips** +and ResourceWarning promoted to error, plus **220/220 Bash assertions**. + ## Exact next slice -Add the schema-10 outcome ledger and `learning.py`: strict final/late records, -append-only corrections, attempt/lead contribution attribution, selection-mode -separation and comparison reports with sample sizes and missingness. Prove that -a failed original attempt later repaired by another profile yields final task -success without crediting the failed attempt, and that a late escaped defect -updates history without erasing the original verdict. +Add `squad learn propose --project PATH`: derive a local, reviewable proposal +packet from the comparison report and latest frozen experiment. It must retain +sample sizes, missingness, every evaluation failure, evidence hashes and the +rollback target; insufficient or absent evidence must produce `no_change` and +must not mutate routing. Then implement versioned profile lifecycle bindings, +qualification and guarded promotion/rollback. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 4c9d6b6..b6ebabf 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -141,6 +141,16 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne observe --file FILE` and frozen/current status evidence are wired. The gate is 255 core tests with 2 optional-SDK skips and 220 Bash assertions. See [M6-STATUS.md](M6-STATUS.md). +- M6 outcome/report/experiment evaluation is verified through `edfb3f3`. + Schema 10 records append-only final and late outcomes with truthful attempt + contribution, and reports separate automatic, pinned and experimental + evidence with sample size and missingness. Schema 11 freezes one-variable + paired experiments, evaluates evaluation and held-out splits, retains every + failure and rollback target, rejects conflicting replay and never changes + active policy. The public path is `squad policy evaluate --experiment FILE`. + The exact gate is 263 core tests with 2 optional-SDK skips and 220 Bash + assertions. Draft `learn propose`, lifecycle bindings/qualification and the + default-off decision helper remain open. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -167,7 +177,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 255 discovered through M6 capacity routing; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | +| Python core discovery | 263 discovered through M6 experiment evaluation; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -204,11 +214,11 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M6 with the outcome/learning ledger: final and late corrections, - attempt/lead contribution, selection-mode separation, comparable reports - and one-variable proposal/rollback evidence. Then add lifecycle - qualification/guarded promotion and the default-off decision helper. Do not - rebuild the completed M5 offline path or capacity ledger. +2. Continue M6 with `learn propose`, keeping insufficient evidence as an + explicit no-change and active routing untouched. Then add lifecycle + qualification/guarded promotion and rollback, followed by the default-off + decision helper. Do not rebuild the completed M5 offline path, capacity + ledger, outcome ledger or experiment evaluator. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 794cd4a..e6b02b2 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -325,6 +325,15 @@ "recorded_at": "2026-09-27T08:41:45-07:00", "availability": "tracked_tests" }, + { + "kind": "learning_evaluation_checkpoint", + "revision": "edfb3f3", + "command_or_action": "263 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, replay-safe outcome/experiment persistence and public report/evaluate CLI tests", + "outcome": "Schemas 10 and 11 preserve final and late outcomes, truthful attempt contribution, selection-mode-separated reports and immutable paired evaluation/held-out experiments. Missing or failed evidence produces no-change, qualifying evidence produces only a promotion proposal with rollback, conflicting replay is fenced and evaluation never mutates active policy.", + "artifact": "M6-STATUS.md", + "recorded_at": "2026-09-29T08:42:48Z", + "availability": "tracked_tests" + }, { "kind": "early_experiment_checkpoint", "revision": "70e59cb", From 98c6685da5ff9a3d76148a31f8857f73010e83fa Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 01:50:03 -0700 Subject: [PATCH 103/197] feat: generate reviewable learning proposals --- plugin/core/src/devsquad/cli.py | 11 ++ plugin/core/src/devsquad/learning.py | 200 +++++++++++++++++++++++++++ plugin/core/src/devsquad/service.py | 68 +++++++++ plugin/core/src/devsquad/store.py | 41 +++++- test/core/test_cli.py | 18 +++ test/core/test_learning.py | 75 +++++++++- test/core/test_service.py | 10 ++ 7 files changed, 421 insertions(+), 2 deletions(-) diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index ce3061a..8859f36 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -178,6 +178,10 @@ def command_policy_evaluate(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).policy_evaluate(experiment)), 0 +def command_learn_propose(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).learning_propose(args.project)), 0 + + def command_status(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).status(args.run)), 0 def command_events(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).events(args.run, args.after, args.limit)), 0 def command_result(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).result(args.run)), 0 @@ -294,6 +298,13 @@ def parser() -> argparse.ArgumentParser: evaluate.add_argument("--json", action="store_true") evaluate.add_argument("--runtime-dir", default=runtime_default) evaluate.set_defaults(func=command_policy_evaluate) + learn = sub.add_parser("learn") + learn_sub = learn.add_subparsers(dest="learn_command", required=True) + propose = learn_sub.add_parser("propose") + propose.add_argument("--project", required=True) + propose.add_argument("--json", action="store_true") + propose.add_argument("--runtime-dir", default=runtime_default) + propose.set_defaults(func=command_learn_propose) handoff = sub.add_parser("handoff") handoff_sub = handoff.add_subparsers(dest="handoff_command", required=True) claim = handoff_sub.add_parser("claim") diff --git a/plugin/core/src/devsquad/learning.py b/plugin/core/src/devsquad/learning.py index c9aa76a..0f1ca12 100644 --- a/plugin/core/src/devsquad/learning.py +++ b/plugin/core/src/devsquad/learning.py @@ -29,6 +29,15 @@ "evidence_availability", "variable", "cases", "gate", "budget", "rollback_target", } +EXPERIMENT_EVALUATION_FIELDS = { + "schema_version", "experiment_id", "spec_sha256", "evaluated_at", + "verdict", "active_policy_changed", "reasons", "metrics", "cases", + "failures", "variable", "rollback_target", "evidence_availability", +} +EXPERIMENT_RECORD_FIELDS = { + "experiment", "spec_sha256", "evaluation", "evaluation_sha256", + "recorded_at", +} def _now(value: datetime | None) -> datetime: @@ -563,3 +572,194 @@ def evaluate_experiment( "rollback_target": spec["rollback_target"], "evidence_availability": spec["evidence_availability"], } + + +def build_learning_proposal( + report: dict[str, Any], + experiment_record: dict[str, Any] | None, + *, + generated_at: str, +) -> dict[str, Any]: + """Distill saved evidence into a reviewable draft without changing policy.""" + _timestamp(generated_at, "generated_at") + required_report_fields = { + "schema_version", "project_id", "project_path", "generated_at", + "sample_size", "terminal_run_count", "final_successes", + "escaped_defects", "lead_repairs", "selection_modes", "profiles", + "missingness", "interpretation", + } + if not isinstance(report, dict) or set(report) != required_report_fields: + raise ContractError("learning proposal report is invalid") + if (report["schema_version"] != 1 + or not isinstance(report["project_path"], str) + or not isinstance(report["selection_modes"], dict) + or not isinstance(report["missingness"], dict)): + raise ContractError("learning proposal report values are invalid") + for field in ("sample_size", "terminal_run_count"): + if type(report[field]) is not int or report[field] < 0: + raise ContractError("learning proposal report counts are invalid") + mode_samples = {} + for mode in sorted(SELECTION_MODES): + row = report["selection_modes"].get(mode) + if (not isinstance(row, dict) or type(row.get("sample_size")) is not int + or row["sample_size"] < 0): + raise ContractError("learning proposal selection samples are invalid") + mode_samples[mode] = row["sample_size"] + + report_sha256 = hashlib.sha256(canonical_json(report).encode()).hexdigest() + experiment_id = None + question = None + hypothesis = None + variable = None + rollback_target = None + failures = [] + reasons = ["no_evaluated_experiment"] + verdict = "no_change" + evidence_availability = "unavailable" + experiment_samples = None + experiment_evidence = None + if experiment_record is not None: + if (not isinstance(experiment_record, dict) + or set(experiment_record) != EXPERIMENT_RECORD_FIELDS): + raise ContractError("learning proposal experiment record is invalid") + spec = validate_experiment(experiment_record["experiment"]) + evaluation = experiment_record["evaluation"] + if (not isinstance(evaluation, dict) + or set(evaluation) != EXPERIMENT_EVALUATION_FIELDS): + raise ContractError("learning proposal evaluation is invalid") + spec_sha256 = hashlib.sha256(canonical_json(spec).encode()).hexdigest() + evaluation_sha256 = hashlib.sha256( + canonical_json(evaluation).encode(), + ).hexdigest() + if (experiment_record["spec_sha256"] != spec_sha256 + or evaluation.get("spec_sha256") != spec_sha256 + or experiment_record["evaluation_sha256"] != evaluation_sha256): + raise ContractError("learning proposal evidence hash is invalid") + if (evaluation.get("experiment_id") != spec["experiment_id"] + or evaluation.get("verdict") not in { + "no_change", "promotion_proposal", + } + or evaluation.get("active_policy_changed") is not False + or evaluation.get("variable") != spec["variable"] + or evaluation.get("rollback_target") != spec["rollback_target"] + or not isinstance(evaluation.get("reasons"), list) + or not isinstance(evaluation.get("failures"), list) + or not isinstance(evaluation.get("metrics"), dict)): + raise ContractError("learning proposal evaluation values are invalid") + _timestamp(experiment_record["recorded_at"], "recorded_at") + metrics = evaluation["metrics"] + if set(metrics) != {"evaluation", "held_out"} or any( + not isinstance(metrics.get(split), dict) + or type(metrics[split].get("declared_pairs")) is not int + or type(metrics[split].get("available_pairs")) is not int + for split in ("evaluation", "held_out")): + raise ContractError("learning proposal experiment samples are invalid") + experiment_id = spec["experiment_id"] + question = spec["question"] + hypothesis = spec["hypothesis"] + variable = spec["variable"] + rollback_target = spec["rollback_target"] + failures = list(evaluation["failures"]) + reasons = list(evaluation["reasons"]) + verdict = evaluation["verdict"] + evidence_availability = spec["evidence_availability"] + experiment_samples = { + split: { + "declared_pairs": metrics[split]["declared_pairs"], + "available_pairs": metrics[split]["available_pairs"], + } + for split in ("evaluation", "held_out") + } + experiment_evidence = { + "experiment_id": experiment_id, + "spec_sha256": spec_sha256, + "evaluation_sha256": evaluation_sha256, + "recorded_at": experiment_record["recorded_at"], + } + + identity = { + "project_path": report["project_path"], + "report_sha256": report_sha256, + "experiment": experiment_evidence, + "verdict": verdict, + } + proposal_id = "proposal-" + hashlib.sha256( + canonical_json(identity).encode(), + ).hexdigest()[:24] + return { + "schema_version": 1, + "proposal_id": proposal_id, + "project_id": report["project_id"], + "project_path": report["project_path"], + "generated_at": generated_at, + "verdict": verdict, + "active_policy_changed": False, + "question": question, + "hypothesis": hypothesis, + "variable": variable, + "rollback_target": rollback_target, + "sample_sizes": { + "terminal_runs": report["terminal_run_count"], + "final_outcomes": report["sample_size"], + "selection_modes": mode_samples, + "experiment": experiment_samples, + }, + "missingness": json.loads(canonical_json(report["missingness"])), + "reasons": reasons, + "failures": failures, + "evidence": { + "availability": evidence_availability, + "report_sha256": report_sha256, + "experiment": experiment_evidence, + }, + "decision": { + "action": ( + "review_policy_change" + if verdict == "promotion_proposal" + else "retain_current_policy" + ), + "review_required": verdict == "promotion_proposal", + }, + } + + +def render_learning_proposal_markdown(proposal: dict[str, Any]) -> str: + """Render a compact local review record for a validated proposal.""" + if not isinstance(proposal, dict) or proposal.get("schema_version") != 1: + raise ContractError("learning proposal is invalid") + lines = [ + f"# Learning proposal {proposal['proposal_id']}", + "", + f"- Verdict: `{proposal['verdict']}`", + f"- Project: `{proposal['project_path']}`", + f"- Generated: `{proposal['generated_at']}`", + "- Active policy changed: `false`", + f"- Next action: `{proposal['decision']['action']}`", + "", + "## Evidence", + "", + f"- Report SHA256: `{proposal['evidence']['report_sha256']}`", + f"- Final outcomes: {proposal['sample_sizes']['final_outcomes']}", + f"- Terminal runs: {proposal['sample_sizes']['terminal_runs']}", + ] + experiment = proposal["evidence"]["experiment"] + if experiment is None: + lines.append("- Experiment: none") + else: + lines.extend([ + f"- Experiment: `{experiment['experiment_id']}`", + f"- Evaluation SHA256: `{experiment['evaluation_sha256']}`", + f"- Rollback target: `{proposal['rollback_target']['profile_id']}` " + f"binding version {proposal['rollback_target']['binding_version']}", + ]) + lines.extend(["", "## Reasons", ""]) + lines.extend( + [f"- `{reason}`" for reason in proposal["reasons"]] + or ["- No gate failures were recorded."] + ) + lines.extend(["", "## Recorded failures", ""]) + lines.extend( + [f"- `{canonical_json(failure)}`" for failure in proposal["failures"]] + or ["- None."] + ) + return "\n".join(lines) + "\n" diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 8a7b788..011bb0b 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -814,6 +814,74 @@ def policy_evaluate(self, experiment: dict[str, Any]) -> dict[str, Any]: finally: store.close() + @staticmethod + def _finalize_learning_file( + directory: Path, name: str, content: bytes, + ) -> dict[str, Any]: + digest = hashlib.sha256(content).hexdigest() + component = Path(name) + if component.name != name or not component.stem or not component.suffix: + raise ContractError("learning file name is invalid") + destination = directory / f"{component.stem}.{digest}{component.suffix}" + descriptor, temporary = tempfile.mkstemp(prefix=f".{name}.", dir=directory) + try: + with os.fdopen(descriptor, "wb") as stream: + stream.write(content) + stream.flush() + os.fsync(stream.fileno()) + try: + os.link(temporary, destination) + except FileExistsError: + if destination.read_bytes() != content: + raise ConflictError("content-addressed learning file is corrupt") + directory_fd = os.open(directory, os.O_RDONLY) + try: + os.fsync(directory_fd) + finally: + os.close(directory_fd) + finally: + if os.path.exists(temporary): + os.unlink(temporary) + return { + "path": str(destination), + "sha256": digest, + "byte_size": len(content), + } + + def learning_propose(self, project: str | Path) -> dict[str, Any]: + """Write a local review draft from saved evidence without policy mutation.""" + from .learning import ( + build_learning_proposal, + render_learning_proposal_markdown, + ) + + if not isinstance(project, (str, Path)): + raise ContractError("proposal project path is invalid") + store = self._store() + try: + inputs = store.learning_proposal_inputs(Path(project)) + finally: + store.close() + generated_at = inputs["report"]["generated_at"] + proposal = build_learning_proposal( + inputs["report"], inputs["experiment"], generated_at=generated_at, + ) + json_content = (canonical_json(proposal) + "\n").encode() + markdown_content = render_learning_proposal_markdown(proposal).encode() + directory = self.runtime / "learning" / "proposals" + directory.mkdir(parents=True, exist_ok=True) + return { + "proposal": proposal, + "artifacts": { + "json": self._finalize_learning_file( + directory, f"{proposal['proposal_id']}.json", json_content, + ), + "markdown": self._finalize_learning_file( + directory, f"{proposal['proposal_id']}.md", markdown_content, + ), + }, + } + @staticmethod def _status_capacity(store: Store, run: dict[str, Any]) -> dict[str, Any] | None: try: diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 6822c11..3598f20 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1189,10 +1189,12 @@ def evaluate_learning_experiment( from .learning import evaluate_experiment, validate_experiment spec = validate_experiment(experiment) + project_path = Path(spec["project_path"]).resolve(strict=True) + spec = {**spec, "project_path": str(project_path)} spec_json = canonical_json(spec) spec_sha256 = hashlib.sha256(spec_json.encode()).hexdigest() current = _authoritative_now(now) - common_dir = git_common_dir(Path(spec["project_path"])) + common_dir = git_common_dir(project_path) self.connection.execute("BEGIN IMMEDIATE") try: existing = self.connection.execute( @@ -1269,6 +1271,43 @@ def evaluate_learning_experiment( self.connection.execute("ROLLBACK") raise + def learning_proposal_inputs( + self, project: Path, *, now: datetime | None = None, + ) -> dict[str, Any]: + """Read a consistent report and the latest frozen project experiment.""" + project_path = project.resolve(strict=True) + current = _authoritative_now(now) + self.connection.execute("BEGIN") + try: + report = self.learning_report(project_path, now=current) + if report["project_id"] is None: + row = self.connection.execute( + "SELECT spec_json,spec_sha256,evaluation_json,evaluation_sha256," + "recorded_at FROM experiments WHERE project_path=? " + "ORDER BY recorded_at DESC,id DESC LIMIT 1", + (str(project_path),), + ).fetchone() + else: + row = self.connection.execute( + "SELECT spec_json,spec_sha256,evaluation_json,evaluation_sha256," + "recorded_at FROM experiments WHERE project_id=? OR " + "(project_id IS NULL AND project_path=?) " + "ORDER BY recorded_at DESC,id DESC LIMIT 1", + (report["project_id"], str(project_path)), + ).fetchone() + experiment = None if row is None else { + "experiment": json.loads(row["spec_json"]), + "spec_sha256": row["spec_sha256"], + "evaluation": json.loads(row["evaluation_json"]), + "evaluation_sha256": row["evaluation_sha256"], + "recorded_at": row["recorded_at"], + } + self.connection.execute("COMMIT") + return {"report": report, "experiment": experiment} + except Exception: + self.connection.execute("ROLLBACK") + raise + def worker_invocations(self, run_id: str) -> int: """Count attempts whose durable runner actually crossed the launch fence.""" return self.connection.execute( diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 761295a..d589c54 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -176,6 +176,24 @@ def test_policy_evaluate_dispatches_frozen_experiment(self): self.assert_success_envelope(payload, response) service.policy_evaluate.assert_called_once_with(experiment) + def test_learn_propose_dispatches_project(self): + service = mock.Mock() + response = { + "proposal": { + "verdict": "no_change", + "active_policy_changed": False, + }, + "artifacts": {}, + } + service.learning_propose.return_value = response + code, payload, stderr = self.invoke([ + "learn", "propose", "--project", str(self.root), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.learning_propose.assert_called_once_with(str(self.root)) + def test_handoff_claim_renew_and_complete_dispatch_parsed_objects(self): claim_payload = { "schema_version": 1, diff --git a/test/core/test_learning.py b/test/core/test_learning.py index d07ac4b..0463752 100644 --- a/test/core/test_learning.py +++ b/test/core/test_learning.py @@ -1,5 +1,6 @@ import copy from datetime import datetime, timedelta, timezone +import hashlib import json from pathlib import Path import subprocess @@ -12,11 +13,14 @@ from devsquad.contracts import ContractError from devsquad.learning import ( + build_comparison_report, + build_learning_proposal, evaluate_experiment, + render_learning_proposal_markdown, validate_experiment, validate_outcome, ) -from devsquad.store import ConflictError, Store +from devsquad.store import ConflictError, Store, canonical_json NOW = datetime(2026, 9, 27, 16, 0, tzinfo=timezone.utc) @@ -236,6 +240,75 @@ def test_experiment_evaluation_is_persisted_and_replay_safe(self): changed["hypothesis"] = "Mutated after evaluation." with self.assertRaisesRegex(ConflictError, "different specification"): store.evaluate_learning_experiment(changed, now=NOW) + inputs = store.learning_proposal_inputs(repository, now=NOW) + self.assertEqual( + inputs["experiment"]["experiment"]["experiment_id"], + spec["experiment_id"], + ) + + def test_learning_proposal_is_traceable_and_never_changes_policy(self): + project_path = "/tmp/experiment-project" + report = build_comparison_report( + project_id=None, + project_path=project_path, + terminal_runs=[], + outcome_records=[], + attempt_profiles={}, + generated_at=NOW.isoformat(), + ) + no_evidence = build_learning_proposal( + report, None, generated_at=NOW.isoformat(), + ) + self.assertEqual(no_evidence["verdict"], "no_change") + self.assertFalse(no_evidence["active_policy_changed"]) + self.assertEqual(no_evidence["reasons"], ["no_evaluated_experiment"]) + self.assertEqual( + no_evidence["decision"], + {"action": "retain_current_policy", "review_required": False}, + ) + + spec = experiment(Path(project_path)) + chains = {} + for case in spec["cases"]: + chains[case["control_outcome_id"]] = { + "final": experimental_final(case["control_outcome_id"], "failed"), + "late_corrections": [], + } + chains[case["candidate_outcome_id"]] = { + "final": experimental_final(case["candidate_outcome_id"], "succeeded"), + "late_corrections": [], + } + evaluation = evaluate_experiment( + spec, chains, evaluated_at=NOW.isoformat(), + ) + spec_sha256 = hashlib.sha256( + canonical_json(validate_experiment(spec)).encode(), + ).hexdigest() + evaluation_sha256 = hashlib.sha256( + canonical_json(evaluation).encode(), + ).hexdigest() + record = { + "experiment": spec, + "spec_sha256": spec_sha256, + "evaluation": evaluation, + "evaluation_sha256": evaluation_sha256, + "recorded_at": NOW.isoformat(), + } + proposal = build_learning_proposal( + report, record, generated_at=NOW.isoformat(), + ) + self.assertEqual(proposal["verdict"], "promotion_proposal") + self.assertFalse(proposal["active_policy_changed"]) + self.assertEqual(proposal["rollback_target"]["profile_id"], "profile-a") + self.assertEqual( + proposal["evidence"]["experiment"]["evaluation_sha256"], + evaluation_sha256, + ) + self.assertIn(proposal["proposal_id"], render_learning_proposal_markdown(proposal)) + tampered = copy.deepcopy(record) + tampered["evaluation_sha256"] = "0" * 64 + with self.assertRaisesRegex(ContractError, "evidence hash"): + build_learning_proposal(report, tampered, generated_at=NOW.isoformat()) def test_final_and_late_outcomes_are_append_only_and_attempt_bound(self): with tempfile.TemporaryDirectory() as root: diff --git a/test/core/test_service.py b/test/core/test_service.py index 6199e8f..85126bf 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -191,6 +191,16 @@ def test_outcome_add_and_project_report_share_saved_ledger(self): self.assertEqual(report["terminal_run_count"], 1) self.assertEqual(report["final_successes"], 1) self.assertEqual(report["missingness"]["finals_without_contributions"], 1) + proposed = self.service.learning_propose(self.repo) + self.assertEqual(proposed["proposal"]["verdict"], "no_change") + self.assertFalse(proposed["proposal"]["active_policy_changed"]) + self.assertEqual( + proposed["proposal"]["reasons"], ["no_evaluated_experiment"], + ) + for artifact in proposed["artifacts"].values(): + content = Path(artifact["path"]).read_bytes() + self.assertEqual(hashlib.sha256(content).hexdigest(), artifact["sha256"]) + self.assertEqual(len(content), artifact["byte_size"]) def test_abandoned_preparation_is_reclaimed_from_the_submitted_request(self): store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") From 3ec12d6b0ad237e143f0276987ca9b58da29e655 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 01:51:01 -0700 Subject: [PATCH 104/197] docs: checkpoint M6 learning proposals --- docs/plans/engineering-team/M6-STATUS.md | 23 ++++++++++++++++------- docs/plans/engineering-team/RESUME.md | 21 ++++++++++++--------- docs/plans/engineering-team/backlog.json | 9 +++++++++ 3 files changed, 37 insertions(+), 16 deletions(-) diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md index eca76e6..46d85a6 100644 --- a/docs/plans/engineering-team/M6-STATUS.md +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -14,7 +14,8 @@ default-off decision helper remain open. | Final and late outcomes | Preserve attempt contribution, lead repair, final success and escaped-defect corrections without crediting failed attempts | verified at `d621df2` | | Comparison reports | Sample sizes, missingness and separated automatic/pinned/experimental evidence | verified at `45ebc9e` | | Frozen experiment evaluation | One-variable paired evaluation/held-out cases, failure evidence, no-change or promotion-proposal verdict and rollback target; evaluation never changes active policy | verified at `edfb3f3` | -| Draft proposals and held-out rerun | `learn propose`, review packet and post-change held-out/rollback evidence | pending | +| Draft proposals | `learn propose` emits content-addressed JSON/Markdown with hashes, sample sizes, missingness, failures and rollback; no evidence yields no-change | verified at `98c6685` | +| Held-out rerun and rollback | Post-change held-out evidence and exercised rollback through lifecycle bindings | pending | | Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | pending | | Decision helper M6-D1 | Default-off typed contract, fake adapter, cache/accounting and authority/integrity tests | in progress; synthetic Jev probe mechanics only | | Jev M6-D2 | One capped synthetic request with exact model/usage/latency/cost receipt | blocked on `TYPESAFE_API_KEY` | @@ -64,11 +65,19 @@ different specification conflicts. The combined checkpoint gate is **263 core tests with 2 optional-SDK skips** and ResourceWarning promoted to error, plus **220/220 Bash assertions**. +`98c6685` adds `squad learn propose --project PATH`. It reads one consistent +ledger snapshot, verifies the latest experiment and evaluation hashes, and +writes local content-addressed JSON and Markdown drafts under the runtime +directory. The draft includes selection-mode sample sizes, missingness, all +recorded evaluation failures and the rollback version. Absent experiment +evidence produces an explicit `no_change`; even qualifying evidence produces +only `promotion_proposal`, with `active_policy_changed: false`. The gate is +**265 core tests with 2 optional-SDK skips** plus **220/220 Bash assertions**. + ## Exact next slice -Add `squad learn propose --project PATH`: derive a local, reviewable proposal -packet from the comparison report and latest frozen experiment. It must retain -sample sizes, missingness, every evaluation failure, evidence hashes and the -rollback target; insufficient or absent evidence must produce `no_change` and -must not mutate routing. Then implement versioned profile lifecycle bindings, -qualification and guarded promotion/rollback. +Implement versioned profile lifecycle bindings, qualification and guarded +promotion/rollback: allowed templates, bounded trials, compare-and-swap +binding versions, new-run-only effects, qualified fallback and immutable +decision receipts. Then exercise a held-out rerun and rollback before wiring +the default-off decision helper. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index b6ebabf..3022d96 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -148,9 +148,12 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne paired experiments, evaluates evaluation and held-out splits, retains every failure and rollback target, rejects conflicting replay and never changes active policy. The public path is `squad policy evaluate --experiment FILE`. - The exact gate is 263 core tests with 2 optional-SDK skips and 220 Bash - assertions. Draft `learn propose`, lifecycle bindings/qualification and the - default-off decision helper remain open. + `98c6685` adds `squad learn propose --project PATH`, which writes + content-addressed local JSON/Markdown drafts from one consistent ledger + snapshot, verifies saved evidence hashes and returns explicit no-change when + evidence is absent. The exact gate is 265 core tests with 2 optional-SDK + skips and 220 Bash assertions. Lifecycle bindings/qualification, held-out + rollback and the default-off decision helper remain open. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -177,7 +180,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 263 discovered through M6 experiment evaluation; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | +| Python core discovery | 265 discovered through M6 learning proposals; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -214,11 +217,11 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M6 with `learn propose`, keeping insufficient evidence as an - explicit no-change and active routing untouched. Then add lifecycle - qualification/guarded promotion and rollback, followed by the default-off - decision helper. Do not rebuild the completed M5 offline path, capacity - ledger, outcome ledger or experiment evaluator. +2. Continue M6 with lifecycle qualification, compare-and-swap reviewed or + guarded promotion, new-run-only bindings and rollback receipts. Exercise a + held-out rerun/rollback, then add the default-off decision helper. Do not + rebuild the completed M5 offline path, capacity ledger, outcome ledger, + experiment evaluator or proposal generator. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index e6b02b2..5504a25 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -334,6 +334,15 @@ "recorded_at": "2026-09-29T08:42:48Z", "availability": "tracked_tests" }, + { + "kind": "learning_proposal_checkpoint", + "revision": "98c6685", + "command_or_action": "265 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions and focused pure/store/service/filesystem/CLI proposal tests", + "outcome": "squad learn propose writes content-addressed local JSON and Markdown from a consistent ledger snapshot, verifies report/experiment/evaluation evidence, preserves sample size, missingness, failures and rollback, returns explicit no-change without evidence and never changes active policy.", + "artifact": "M6-STATUS.md", + "recorded_at": "2026-09-29T08:50:09Z", + "availability": "tracked_tests" + }, { "kind": "early_experiment_checkpoint", "revision": "70e59cb", From c0ce7433e6f8c4e5673524c9c1de9879e7b64305 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:03:13 -0700 Subject: [PATCH 105/197] feat: persist qualified profile lifecycle --- plugin/core/src/devsquad/lifecycle.py | 339 +++++++++++ .../migrations/012_profile_lifecycle.sql | 87 +++ plugin/core/src/devsquad/store.py | 534 +++++++++++++++++- test/core/test_capacity.py | 2 +- test/core/test_cli.py | 4 +- test/core/test_handoff_store.py | 6 +- test/core/test_learning.py | 2 +- test/core/test_lifecycle.py | 355 ++++++++++++ test/core/test_store.py | 8 +- 9 files changed, 1325 insertions(+), 12 deletions(-) create mode 100644 plugin/core/src/devsquad/lifecycle.py create mode 100644 plugin/core/src/devsquad/migrations/012_profile_lifecycle.sql create mode 100644 test/core/test_lifecycle.py diff --git a/plugin/core/src/devsquad/lifecycle.py b/plugin/core/src/devsquad/lifecycle.py new file mode 100644 index 0000000..7618dfe --- /dev/null +++ b/plugin/core/src/devsquad/lifecycle.py @@ -0,0 +1,339 @@ +"""Strict M6 profile lifecycle contracts and qualification gates.""" + +from __future__ import annotations + +from datetime import datetime +import hashlib +import json +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .validation import validate_profile + + +TEMPLATE_FIELDS = { + "schema_version", "template_id", "alias", "update_mode", "policy", + "allowed_harnesses", "allowed_model_families", "allowed_account_pools", + "allowed_task_classes", "permission_policy", "allowed_tools", + "allowed_billing_modes", "gate", +} +QUALIFICATION_FIELDS = { + "schema_version", "qualification_id", "alias", "template_id", + "candidate_profile", "task_class", "experiment_id", "evaluation_sha256", + "source", "budget", "measured", "verdict", "evidence_refs", +} +BINDING_CHANGE_FIELDS = { + "schema_version", "decision_id", "action", "alias", + "expected_binding_version", "qualification_id", "rollback_target", + "actor", "reason", "evidence_refs", +} +SHA256_LENGTH = 64 + + +def _exact(value: Any, fields: set[str], label: str) -> dict[str, Any]: + if not isinstance(value, dict) or set(value) != fields: + raise ContractError(f"{label} fields are invalid") + return value + + +def _identifier(value: Any, field: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ContractError(f"lifecycle {field} must be a non-empty string") + return value + + +def _strings(value: Any, field: str, *, required: bool = True) -> list[str]: + if (not isinstance(value, list) + or (required and not value) + or any(not isinstance(item, str) or not item for item in value) + or len(set(value)) != len(value)): + raise ContractError(f"lifecycle {field} must contain unique strings") + return list(value) + + +def _sha256(value: Any, field: str, *, nullable: bool = False) -> str | None: + if value is None and nullable: + return None + if (not isinstance(value, str) or len(value) != SHA256_LENGTH + or any(character not in "0123456789abcdef" for character in value)): + raise ContractError(f"lifecycle {field} must be a SHA256 digest") + return value + + +def _timestamp(value: Any, field: str) -> str: + if not isinstance(value, str) or not value: + raise ContractError(f"lifecycle {field} must be a timestamp") + try: + parsed = datetime.fromisoformat(value) + except ValueError as exc: + raise ContractError(f"lifecycle {field} must be an ISO timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ContractError(f"lifecycle {field} must include a timezone") + return value + + +def validate_profile_template(value: dict[str, Any]) -> dict[str, Any]: + """Validate one reviewed, versioned alias qualification policy.""" + _exact(value, TEMPLATE_FIELDS, "profile template") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("profile template schema_version is invalid") + template_id = _identifier(value["template_id"], "template_id") + alias = _identifier(value["alias"], "alias") + if value["update_mode"] not in {"reviewed", "guarded_auto"}: + raise ContractError("profile template update_mode is invalid") + policy = _exact(value["policy"], {"id", "version"}, "profile template policy") + _identifier(policy["id"], "policy.id") + if type(policy["version"]) is not int or policy["version"] < 1: + raise ContractError("profile template policy version is invalid") + permission = value["permission_policy"] + if permission not in {"read_only", "workspace_write"}: + raise ContractError("profile template permission policy is invalid") + billing = _strings(value["allowed_billing_modes"], "allowed_billing_modes") + if any(mode not in {"subscription", "paid_api"} for mode in billing): + raise ContractError("profile template billing mode is invalid") + gate = _exact(value["gate"], { + "min_evaluation_pairs", "min_held_out_pairs", "max_critical_defects", + "max_latency_ratio", "max_usage_ratio", + }, "profile template gate") + for field in ( + "min_evaluation_pairs", "min_held_out_pairs", "max_critical_defects", + ): + if type(gate[field]) is not int or gate[field] < (0 if field.startswith("max_") else 1): + raise ContractError(f"profile template gate {field} is invalid") + for field in ("max_latency_ratio", "max_usage_ratio"): + number = gate[field] + if (number is not None and (isinstance(number, bool) + or not isinstance(number, (int, float)) or number <= 0)): + raise ContractError(f"profile template gate {field} is invalid") + normalized = { + **value, + "template_id": template_id, + "alias": alias, + "policy": dict(policy), + "allowed_harnesses": _strings( + value["allowed_harnesses"], "allowed_harnesses", + ), + "allowed_model_families": _strings( + value["allowed_model_families"], "allowed_model_families", + ), + "allowed_account_pools": _strings( + value["allowed_account_pools"], "allowed_account_pools", + ), + "allowed_task_classes": _strings( + value["allowed_task_classes"], "allowed_task_classes", + ), + "allowed_tools": _strings( + value["allowed_tools"], "allowed_tools", required=False, + ), + "allowed_billing_modes": billing, + "gate": dict(gate), + } + return json.loads(canonical_json(normalized)) + + +def profile_fingerprint(profile: dict[str, Any]) -> str: + validate_profile(profile) + return hashlib.sha256(canonical_json(profile).encode()).hexdigest() + + +def profile_template_violation( + profile: dict[str, Any], template: dict[str, Any], +) -> str | None: + """Return the first authority-boundary violation, if any.""" + validate_profile(profile) + normalized = validate_profile_template(template) + checks = ( + (profile["harness"] in normalized["allowed_harnesses"], "harness_not_allowed"), + ( + profile["model_family"] in normalized["allowed_model_families"], + "model_family_not_allowed", + ), + ( + profile["account_pool_id"] in normalized["allowed_account_pools"], + "account_pool_not_allowed", + ), + ( + profile["permission_policy"] == normalized["permission_policy"], + "permission_change_not_allowed", + ), + ( + set(profile["required_tools"]) <= set(normalized["allowed_tools"]), + "tool_not_allowed", + ), + ( + profile["billing_mode"] in normalized["allowed_billing_modes"], + "billing_mode_not_allowed", + ), + ) + return next((reason for allowed, reason in checks if not allowed), None) + + +def guarded_change_violation( + incumbent: dict[str, Any], candidate: dict[str, Any], +) -> str | None: + """Prevent guarded automation from widening account/tool authority.""" + validate_profile(incumbent) + validate_profile(candidate) + if incumbent["permission_policy"] != candidate["permission_policy"]: + return "guarded_permission_change" + if incumbent["billing_mode"] != candidate["billing_mode"]: + return "guarded_billing_change" + if incumbent["account_pool_id"] != candidate["account_pool_id"]: + return "guarded_account_route_change" + if not set(candidate["required_tools"]) <= set(incumbent["required_tools"]): + return "guarded_tool_expansion" + return None + + +def validate_qualification(value: dict[str, Any]) -> dict[str, Any]: + """Validate a bounded candidate qualification record.""" + _exact(value, QUALIFICATION_FIELDS, "qualification") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("qualification schema_version is invalid") + qualification_id = _identifier(value["qualification_id"], "qualification_id") + alias = _identifier(value["alias"], "qualification.alias") + template_id = _identifier(value["template_id"], "qualification.template_id") + task_class = _identifier(value["task_class"], "qualification.task_class") + candidate = json.loads(canonical_json(value["candidate_profile"])) + validate_profile(candidate) + experiment_id = value["experiment_id"] + evaluation_sha256 = value["evaluation_sha256"] + if (experiment_id is None) != (evaluation_sha256 is None): + raise ContractError("qualification experiment evidence must be paired") + if experiment_id is not None: + _identifier(experiment_id, "qualification.experiment_id") + _sha256(evaluation_sha256, "qualification.evaluation_sha256") + source = _exact(value["source"], { + "harness_version", "catalog_sha256", "model_revision", + }, "qualification source") + if source["harness_version"] is not None: + _identifier(source["harness_version"], "source.harness_version") + if source["model_revision"] is not None: + _identifier(source["model_revision"], "source.model_revision") + _sha256(source["catalog_sha256"], "source.catalog_sha256", nullable=True) + budget = _exact(value["budget"], { + "max_cases", "used_cases", "max_worker_invocations", + "worker_invocations", "max_wall_seconds", "wall_seconds", + }, "qualification budget") + for field in budget: + if type(budget[field]) is not int or budget[field] < 0: + raise ContractError(f"qualification budget {field} is invalid") + if (budget["used_cases"] > budget["max_cases"] + or budget["worker_invocations"] > budget["max_worker_invocations"] + or budget["wall_seconds"] > budget["max_wall_seconds"]): + raise ContractError("qualification exceeded its frozen budget") + measured = _exact(value["measured"], { + "evaluation_pairs", "held_out_pairs", "critical_defects", + "latency_ratio", "usage_ratio", + }, "qualification measurements") + for field in ("evaluation_pairs", "held_out_pairs", "critical_defects"): + if type(measured[field]) is not int or measured[field] < 0: + raise ContractError(f"qualification measurement {field} is invalid") + for field in ("latency_ratio", "usage_ratio"): + number = measured[field] + if (number is not None and (isinstance(number, bool) + or not isinstance(number, (int, float)) or number < 0)): + raise ContractError(f"qualification measurement {field} is invalid") + if value["verdict"] not in {"qualified", "rejected", "incomplete"}: + raise ContractError("qualification verdict is invalid") + normalized = { + **value, + "qualification_id": qualification_id, + "alias": alias, + "template_id": template_id, + "task_class": task_class, + "candidate_profile": candidate, + "source": dict(source), + "budget": dict(budget), + "measured": dict(measured), + "evidence_refs": _strings( + value["evidence_refs"], "qualification.evidence_refs", + required=value["verdict"] == "qualified", + ), + } + return json.loads(canonical_json(normalized)) + + +def qualification_gate_failures( + qualification: dict[str, Any], template: dict[str, Any], +) -> list[str]: + """Evaluate the static preauthorized qualification gate.""" + record = validate_qualification(qualification) + policy = validate_profile_template(template) + reasons = [] + if record["alias"] != policy["alias"] or record["template_id"] != policy["template_id"]: + reasons.append("template_identity_mismatch") + if record["task_class"] not in policy["allowed_task_classes"]: + reasons.append("task_class_not_allowed") + violation = profile_template_violation(record["candidate_profile"], policy) + if violation: + reasons.append(violation) + if record["candidate_profile"]["quality_status"] not in {"trial", "proven"}: + reasons.append("candidate_quality_not_eligible") + if record["experiment_id"] is None: + reasons.append("experiment_evidence_missing") + measured = record["measured"] + gate = policy["gate"] + if measured["evaluation_pairs"] < gate["min_evaluation_pairs"]: + reasons.append("insufficient_evaluation_pairs") + if measured["held_out_pairs"] < gate["min_held_out_pairs"]: + reasons.append("insufficient_held_out_pairs") + if measured["critical_defects"] > gate["max_critical_defects"]: + reasons.append("critical_defect_limit_exceeded") + for measurement, limit in ( + ("latency_ratio", "max_latency_ratio"), + ("usage_ratio", "max_usage_ratio"), + ): + if gate[limit] is not None and measured[measurement] is None: + reasons.append(f"{measurement}_missing") + elif (gate[limit] is not None + and measured[measurement] > gate[limit]): + reasons.append(f"{measurement}_limit_exceeded") + return sorted(set(reasons)) + + +def validate_binding_change(value: dict[str, Any]) -> dict[str, Any]: + """Validate a replay-safe promotion or rollback request.""" + _exact(value, BINDING_CHANGE_FIELDS, "binding change") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("binding change schema_version is invalid") + decision_id = _identifier(value["decision_id"], "decision_id") + alias = _identifier(value["alias"], "binding change alias") + if value["action"] not in {"promote", "rollback"}: + raise ContractError("binding change action is invalid") + if (type(value["expected_binding_version"]) is not int + or value["expected_binding_version"] < 1): + raise ContractError("binding change expected version is invalid") + qualification_id = value["qualification_id"] + rollback_target = value["rollback_target"] + if value["action"] == "promote": + _identifier(qualification_id, "binding change qualification_id") + if rollback_target is not None: + raise ContractError("promotion cannot specify a rollback target") + else: + if qualification_id is not None: + raise ContractError("rollback cannot specify a qualification") + _exact(rollback_target, {"profile_id", "binding_version"}, "rollback target") + _identifier(rollback_target["profile_id"], "rollback profile_id") + if type(rollback_target["binding_version"]) is not int or rollback_target["binding_version"] < 1: + raise ContractError("rollback binding_version is invalid") + if value["actor"] not in {"human", "guarded_auto"}: + raise ContractError("binding change actor is invalid") + reason = _identifier(value["reason"], "binding change reason") + return json.loads(canonical_json({ + **value, + "decision_id": decision_id, + "alias": alias, + "reason": reason, + "evidence_refs": _strings( + value["evidence_refs"], "binding change evidence_refs", required=True, + ), + "rollback_target": ( + dict(rollback_target) if rollback_target is not None else None + ), + })) + + +def validate_recorded_at(value: Any) -> str: + return _timestamp(value, "recorded_at") diff --git a/plugin/core/src/devsquad/migrations/012_profile_lifecycle.sql b/plugin/core/src/devsquad/migrations/012_profile_lifecycle.sql new file mode 100644 index 0000000..83b7a44 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/012_profile_lifecycle.sql @@ -0,0 +1,87 @@ +CREATE TABLE profile_templates ( + template_id TEXT PRIMARY KEY, + alias TEXT NOT NULL, + update_mode TEXT NOT NULL CHECK(update_mode IN ('reviewed', 'guarded_auto')), + policy_id TEXT NOT NULL, + policy_version INTEGER NOT NULL CHECK(policy_version >= 1), + payload_json TEXT NOT NULL, + payload_sha256 TEXT NOT NULL + CHECK(length(payload_sha256) = 64 AND payload_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL +); + +CREATE INDEX profile_templates_alias_policy +ON profile_templates(alias, policy_id, policy_version, recorded_at); + +CREATE TABLE concrete_profiles ( + profile_id TEXT PRIMARY KEY, + profile_json TEXT NOT NULL, + profile_sha256 TEXT NOT NULL + CHECK(length(profile_sha256) = 64 AND profile_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL +); + +CREATE TABLE qualification_runs ( + qualification_id TEXT PRIMARY KEY, + alias TEXT NOT NULL, + template_id TEXT NOT NULL REFERENCES profile_templates(template_id), + profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + experiment_id TEXT REFERENCES experiments(experiment_id), + evaluation_sha256 TEXT, + verdict TEXT NOT NULL CHECK(verdict IN ('qualified', 'rejected', 'incomplete')), + payload_json TEXT NOT NULL, + payload_sha256 TEXT NOT NULL + CHECK(length(payload_sha256) = 64 AND payload_sha256 NOT GLOB '*[^0-9a-f]*'), + gate_failures_json TEXT NOT NULL, + recorded_at TEXT NOT NULL, + CHECK( + (experiment_id IS NULL AND evaluation_sha256 IS NULL) + OR + (experiment_id IS NOT NULL AND length(evaluation_sha256) = 64 + AND evaluation_sha256 NOT GLOB '*[^0-9a-f]*') + ) +); + +CREATE INDEX qualification_runs_alias_history +ON qualification_runs(alias, recorded_at, qualification_id); + +CREATE TABLE profile_bindings ( + alias TEXT PRIMARY KEY, + template_id TEXT NOT NULL REFERENCES profile_templates(template_id), + profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + qualification_id TEXT REFERENCES qualification_runs(qualification_id), + version INTEGER NOT NULL CHECK(version >= 1), + updated_at TEXT NOT NULL +); + +CREATE TABLE profile_binding_versions ( + alias TEXT NOT NULL, + version INTEGER NOT NULL CHECK(version >= 1), + template_id TEXT NOT NULL REFERENCES profile_templates(template_id), + profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + qualification_id TEXT REFERENCES qualification_runs(qualification_id), + decision_id TEXT, + recorded_at TEXT NOT NULL, + PRIMARY KEY(alias, version) +); + +CREATE TABLE binding_decisions ( + decision_id TEXT PRIMARY KEY, + alias TEXT NOT NULL, + action TEXT NOT NULL CHECK(action IN ('promote', 'rollback')), + actor TEXT NOT NULL CHECK(actor IN ('human', 'guarded_auto')), + from_version INTEGER NOT NULL CHECK(from_version >= 1), + to_version INTEGER NOT NULL CHECK(to_version > from_version), + from_profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + to_profile_id TEXT NOT NULL REFERENCES concrete_profiles(profile_id), + qualification_id TEXT REFERENCES qualification_runs(qualification_id), + request_sha256 TEXT NOT NULL + CHECK(length(request_sha256) = 64 AND request_sha256 NOT GLOB '*[^0-9a-f]*'), + receipt_json TEXT NOT NULL, + receipt_sha256 TEXT NOT NULL + CHECK(length(receipt_sha256) = 64 AND receipt_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL +); + +CREATE INDEX binding_decisions_alias_history +ON binding_decisions(alias, to_version, decision_id); diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 3598f20..0a00f79 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 11 +SUPPORTED_SCHEMA_VERSION = 12 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -1308,6 +1308,538 @@ def learning_proposal_inputs( self.connection.execute("ROLLBACK") raise + @staticmethod + def _profile_record(profile: dict[str, Any]) -> tuple[str, str]: + from .lifecycle import profile_fingerprint + + payload = canonical_json(profile) + return payload, profile_fingerprint(profile) + + def _insert_concrete_profile( + self, profile: dict[str, Any], recorded_at: str, + ) -> tuple[str, str]: + payload, digest = self._profile_record(profile) + existing = self.connection.execute( + "SELECT profile_json,profile_sha256 FROM concrete_profiles " + "WHERE profile_id=?", + (profile["id"],), + ).fetchone() + if existing is not None: + if (existing["profile_json"] != payload + or existing["profile_sha256"] != digest): + raise ConflictError( + "profile id was already used with different concrete settings", + ) + return payload, digest + self.connection.execute( + "INSERT INTO concrete_profiles(profile_id,profile_json,profile_sha256," + "recorded_at) VALUES(?,?,?,?)", + (profile["id"], payload, digest, recorded_at), + ) + return payload, digest + + def _insert_profile_template( + self, template: dict[str, Any], recorded_at: str, + ) -> tuple[str, str]: + payload = canonical_json(template) + digest = hashlib.sha256(payload.encode()).hexdigest() + existing = self.connection.execute( + "SELECT payload_json,payload_sha256 FROM profile_templates " + "WHERE template_id=?", + (template["template_id"],), + ).fetchone() + if existing is not None: + if (existing["payload_json"] != payload + or existing["payload_sha256"] != digest): + raise ConflictError( + "template id was already used with a different policy", + ) + return payload, digest + self.connection.execute( + "INSERT INTO profile_templates(template_id,alias,update_mode,policy_id," + "policy_version,payload_json,payload_sha256,recorded_at) " + "VALUES(?,?,?,?,?,?,?,?)", + ( + template["template_id"], template["alias"], + template["update_mode"], template["policy"]["id"], + template["policy"]["version"], payload, digest, recorded_at, + ), + ) + return payload, digest + + def register_profile_template( + self, template: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """Register an immutable reviewed lifecycle template.""" + from .lifecycle import validate_profile_template + + normalized = validate_profile_template(template) + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + before = self.connection.execute( + "SELECT 1 FROM profile_templates WHERE template_id=?", + (normalized["template_id"],), + ).fetchone() + _, digest = self._insert_profile_template(normalized, recorded_at) + self.connection.execute("COMMIT") + return { + "template": normalized, + "template_sha256": digest, + "recorded_at": recorded_at, + "replayed": before is not None, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def bootstrap_profile_binding( + self, + template: dict[str, Any], + profile: dict[str, Any], + *, + version: int, + now: datetime | None = None, + ) -> dict[str, Any]: + """Import one reviewed baseline binding without fabricating qualification.""" + from .lifecycle import ( + profile_template_violation, + validate_profile_template, + ) + + normalized_template = validate_profile_template(template) + profile_payload, profile_digest = self._profile_record(profile) + profile = json.loads(profile_payload) + if type(version) is not int or version < 1: + raise ContractError("baseline binding version must be positive") + violation = profile_template_violation(profile, normalized_template) + if violation is not None: + raise ContractError(f"baseline profile violates template: {violation}") + recorded_at = _authoritative_now(now).isoformat() + alias = normalized_template["alias"] + self.connection.execute("BEGIN IMMEDIATE") + try: + self._insert_profile_template(normalized_template, recorded_at) + self._insert_concrete_profile(profile, recorded_at) + existing = self.connection.execute( + "SELECT template_id,profile_id,qualification_id,version,updated_at " + "FROM profile_bindings WHERE alias=?", + (alias,), + ).fetchone() + if existing is not None: + if (existing["template_id"] != normalized_template["template_id"] + or existing["profile_id"] != profile["id"] + or existing["qualification_id"] is not None + or existing["version"] != version): + raise ConflictError( + "baseline alias is already bound differently", + ) + self.connection.execute("COMMIT") + return { + "binding": dict(existing), + "profile_sha256": profile_digest, + "replayed": True, + } + self.connection.execute( + "INSERT INTO profile_bindings(alias,template_id,profile_id," + "qualification_id,version,updated_at) VALUES(?,?,?,NULL,?,?)", + ( + alias, normalized_template["template_id"], profile["id"], + version, recorded_at, + ), + ) + self.connection.execute( + "INSERT INTO profile_binding_versions(alias,version,template_id," + "profile_id,qualification_id,decision_id,recorded_at) " + "VALUES(?,?,?,?,NULL,NULL,?)", + ( + alias, version, normalized_template["template_id"], + profile["id"], recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "binding": { + "alias": alias, + "template_id": normalized_template["template_id"], + "profile_id": profile["id"], + "qualification_id": None, + "version": version, + "updated_at": recorded_at, + }, + "profile_sha256": profile_digest, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def record_profile_qualification( + self, qualification: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """Persist bounded qualification evidence after checking saved evaluation.""" + from .lifecycle import ( + qualification_gate_failures, + validate_profile_template, + validate_qualification, + ) + + record = validate_qualification(qualification) + payload = canonical_json(record) + digest = hashlib.sha256(payload.encode()).hexdigest() + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT payload_json,payload_sha256,gate_failures_json,recorded_at " + "FROM qualification_runs WHERE qualification_id=?", + (record["qualification_id"],), + ).fetchone() + if existing is not None: + if existing["payload_json"] != payload: + raise ConflictError( + "qualification id was already used with different evidence", + ) + self.connection.execute("COMMIT") + return { + "qualification": record, + "qualification_sha256": existing["payload_sha256"], + "gate_failures": json.loads(existing["gate_failures_json"]), + "recorded_at": existing["recorded_at"], + "replayed": True, + } + template_row = self.connection.execute( + "SELECT payload_json FROM profile_templates WHERE template_id=?", + (record["template_id"],), + ).fetchone() + if template_row is None: + raise ContractError("qualification template is not registered") + template = validate_profile_template( + json.loads(template_row["payload_json"]), + ) + failures = qualification_gate_failures(record, template) + experiment = None + if record["experiment_id"] is not None: + experiment = self.connection.execute( + "SELECT spec_json,evaluation_json,evaluation_sha256,verdict " + "FROM experiments WHERE experiment_id=?", + (record["experiment_id"],), + ).fetchone() + if (experiment is None + or experiment["evaluation_sha256"] + != record["evaluation_sha256"]): + raise ContractError( + "qualification experiment evidence is unavailable", + ) + spec = json.loads(experiment["spec_json"]) + evaluation = json.loads(experiment["evaluation_json"]) + if (spec["variable"]["alias"] != record["alias"] + or spec["variable"]["candidate_profile_id"] + != record["candidate_profile"]["id"]): + raise ContractError( + "qualification candidate does not match the experiment", + ) + metrics = evaluation["metrics"] + escaped = sum( + metrics[split]["candidate_escaped_defects"] + for split in ("evaluation", "held_out") + ) + if (record["measured"]["evaluation_pairs"] + != metrics["evaluation"]["available_pairs"] + or record["measured"]["held_out_pairs"] + != metrics["held_out"]["available_pairs"] + or record["measured"]["critical_defects"] != escaped): + raise ContractError( + "qualification measurements do not match saved evaluation", + ) + if experiment["verdict"] != "promotion_proposal": + failures.append("experiment_did_not_propose_promotion") + failures = sorted(set(failures)) + if record["verdict"] == "qualified" and failures: + raise ContractError( + "qualification gate did not pass: " + ", ".join(failures), + ) + self._insert_concrete_profile( + record["candidate_profile"], recorded_at, + ) + self.connection.execute( + "INSERT INTO qualification_runs(qualification_id,alias,template_id," + "profile_id,experiment_id,evaluation_sha256,verdict,payload_json," + "payload_sha256,gate_failures_json,recorded_at) " + "VALUES(?,?,?,?,?,?,?,?,?,?,?)", + ( + record["qualification_id"], record["alias"], + record["template_id"], record["candidate_profile"]["id"], + record["experiment_id"], record["evaluation_sha256"], + record["verdict"], payload, digest, + canonical_json(failures), recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "qualification": record, + "qualification_sha256": digest, + "gate_failures": failures, + "recorded_at": recorded_at, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def profile_binding(self, alias: str) -> dict[str, Any] | None: + if not isinstance(alias, str) or not alias: + raise ContractError("profile binding alias is invalid") + row = self.connection.execute( + "SELECT b.alias,b.template_id,b.profile_id,b.qualification_id,b.version," + "b.updated_at,p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256 FROM profile_bindings b " + "JOIN concrete_profiles p ON p.profile_id=b.profile_id " + "JOIN profile_templates t ON t.template_id=b.template_id " + "WHERE b.alias=?", + (alias,), + ).fetchone() + if row is None: + return None + return { + "alias": row["alias"], + "template_id": row["template_id"], + "profile_id": row["profile_id"], + "qualification_id": row["qualification_id"], + "version": row["version"], + "updated_at": row["updated_at"], + "profile": json.loads(row["profile_json"]), + "profile_sha256": row["profile_sha256"], + "template": json.loads(row["template_json"]), + "template_sha256": row["template_sha256"], + } + + def change_profile_binding( + self, change: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """CAS-promote or roll back one alias and persist its decision receipt.""" + from .lifecycle import ( + guarded_change_violation, + validate_binding_change, + ) + + request = validate_binding_change(change) + request_json = canonical_json(request) + request_sha256 = hashlib.sha256(request_json.encode()).hexdigest() + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT request_sha256,receipt_json,receipt_sha256,recorded_at " + "FROM binding_decisions WHERE decision_id=?", + (request["decision_id"],), + ).fetchone() + if existing is not None: + if existing["request_sha256"] != request_sha256: + raise ConflictError( + "decision id was already used with a different request", + ) + self.connection.execute("COMMIT") + return { + "receipt": json.loads(existing["receipt_json"]), + "receipt_sha256": existing["receipt_sha256"], + "recorded_at": existing["recorded_at"], + "replayed": True, + } + current = self.connection.execute( + "SELECT b.template_id,b.profile_id,b.qualification_id,b.version," + "p.profile_json,p.profile_sha256,t.payload_json,t.payload_sha256 " + "FROM profile_bindings b " + "JOIN concrete_profiles p ON p.profile_id=b.profile_id " + "JOIN profile_templates t ON t.template_id=b.template_id " + "WHERE b.alias=?", + (request["alias"],), + ).fetchone() + if current is None: + raise ContractError("profile binding does not exist") + if current["version"] != request["expected_binding_version"]: + raise ConflictError("profile binding version changed") + current_template = json.loads(current["payload_json"]) + current_profile = json.loads(current["profile_json"]) + if (request["actor"] == "guarded_auto" + and current_template["update_mode"] != "guarded_auto"): + raise ContractError( + "guarded automatic promotion is not enabled", + ) + qualification_payload = None + if request["action"] == "promote": + qualification = self.connection.execute( + "SELECT q.alias,q.template_id,q.profile_id,q.verdict," + "q.payload_json,q.payload_sha256,q.gate_failures_json," + "p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256 FROM qualification_runs q " + "JOIN concrete_profiles p ON p.profile_id=q.profile_id " + "JOIN profile_templates t ON t.template_id=q.template_id " + "WHERE q.qualification_id=?", + (request["qualification_id"],), + ).fetchone() + if (qualification is None or qualification["alias"] != request["alias"] + or qualification["verdict"] != "qualified" + or json.loads(qualification["gate_failures_json"])): + raise ContractError( + "binding promotion requires a qualified candidate", + ) + target_template = json.loads(qualification["template_json"]) + target_profile = json.loads(qualification["profile_json"]) + target_template_id = qualification["template_id"] + target_profile_sha256 = qualification["profile_sha256"] + target_template_sha256 = qualification["template_sha256"] + target_qualification_id = request["qualification_id"] + qualification_payload = json.loads(qualification["payload_json"]) + if (request["actor"] == "guarded_auto" + and target_template_id != current["template_id"]): + raise ContractError( + "guarded automation cannot change lifecycle policy", + ) + guarded_violation = ( + guarded_change_violation(current_profile, target_profile) + if request["actor"] == "guarded_auto" else None + ) + if guarded_violation is not None: + raise ContractError(guarded_violation) + else: + rollback = request["rollback_target"] + target = self.connection.execute( + "SELECT v.template_id,v.profile_id,v.qualification_id," + "p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256 FROM profile_binding_versions v " + "JOIN concrete_profiles p ON p.profile_id=v.profile_id " + "JOIN profile_templates t ON t.template_id=v.template_id " + "WHERE v.alias=? AND v.version=?", + (request["alias"], rollback["binding_version"]), + ).fetchone() + if (target is None or target["profile_id"] != rollback["profile_id"] + or rollback["binding_version"] >= current["version"]): + raise ContractError("rollback target is not a prior binding") + target_profile = json.loads(target["profile_json"]) + if target_profile["quality_status"] == "suspended": + raise ContractError("rollback target is suspended") + target_template = json.loads(target["template_json"]) + target_template_id = target["template_id"] + target_profile_sha256 = target["profile_sha256"] + target_template_sha256 = target["template_sha256"] + target_qualification_id = target["qualification_id"] + if target_qualification_id is not None: + qualified = self.connection.execute( + "SELECT verdict,payload_json FROM qualification_runs " + "WHERE qualification_id=?", + (target_qualification_id,), + ).fetchone() + if qualified is None or qualified["verdict"] != "qualified": + raise ContractError("rollback target is no longer qualified") + qualification_payload = json.loads(qualified["payload_json"]) + if (request["actor"] == "guarded_auto" + and target_template_id != current["template_id"]): + raise ContractError( + "guarded automation cannot change lifecycle policy", + ) + if target_profile["id"] == current["profile_id"]: + raise ConflictError("binding already targets the requested profile") + new_version = current["version"] + 1 + receipt = { + "schema_version": 1, + "decision_id": request["decision_id"], + "action": request["action"], + "alias": request["alias"], + "actor": request["actor"], + "reason": request["reason"], + "evidence_refs": request["evidence_refs"], + "from": { + "binding_version": current["version"], + "profile_id": current["profile_id"], + "profile_sha256": current["profile_sha256"], + "template_id": current["template_id"], + "template_sha256": current["payload_sha256"], + }, + "to": { + "binding_version": new_version, + "profile_id": target_profile["id"], + "profile_sha256": target_profile_sha256, + "template_id": target_template_id, + "template_sha256": target_template_sha256, + }, + "qualification_id": target_qualification_id, + "qualification": qualification_payload, + "policy": target_template["policy"], + "rollback_target": { + "profile_id": current["profile_id"], + "binding_version": current["version"], + }, + "requested_rollback_target": ( + request["rollback_target"] + if request["action"] == "rollback" else None + ), + "effective_at": recorded_at, + "affects_new_runs_only": True, + } + receipt_json = canonical_json(receipt) + receipt_sha256 = hashlib.sha256(receipt_json.encode()).hexdigest() + self.connection.execute( + "UPDATE profile_bindings SET template_id=?,profile_id=?," + "qualification_id=?,version=?,updated_at=? WHERE alias=? AND version=?", + ( + target_template_id, target_profile["id"], + target_qualification_id, new_version, recorded_at, + request["alias"], current["version"], + ), + ) + if self.connection.execute("SELECT changes()").fetchone()[0] != 1: + raise ConflictError("profile binding version changed") + self.connection.execute( + "INSERT INTO profile_binding_versions(alias,version,template_id," + "profile_id,qualification_id,decision_id,recorded_at) " + "VALUES(?,?,?,?,?,?,?)", + ( + request["alias"], new_version, target_template_id, + target_profile["id"], target_qualification_id, + request["decision_id"], recorded_at, + ), + ) + self.connection.execute( + "INSERT INTO binding_decisions(decision_id,alias,action,actor," + "from_version,to_version,from_profile_id,to_profile_id," + "qualification_id,request_sha256,receipt_json,receipt_sha256," + "recorded_at) VALUES(?,?,?,?,?,?,?,?,?,?,?,?,?)", + ( + request["decision_id"], request["alias"], request["action"], + request["actor"], current["version"], new_version, + current["profile_id"], target_profile["id"], + target_qualification_id, request_sha256, receipt_json, + receipt_sha256, recorded_at, + ), + ) + self.connection.execute("COMMIT") + return { + "receipt": receipt, + "receipt_sha256": receipt_sha256, + "recorded_at": recorded_at, + "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def profile_binding_decisions(self, alias: str) -> list[dict[str, Any]]: + if not isinstance(alias, str) or not alias: + raise ContractError("profile binding alias is invalid") + return [ + { + "receipt": json.loads(row["receipt_json"]), + "receipt_sha256": row["receipt_sha256"], + "recorded_at": row["recorded_at"], + } + for row in self.connection.execute( + "SELECT receipt_json,receipt_sha256,recorded_at " + "FROM binding_decisions WHERE alias=? ORDER BY to_version", + (alias,), + ) + ] + def worker_invocations(self, run_id: str) -> int: """Count attempts whose durable runner actually crossed the launch fence.""" return self.connection.execute( diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index 7de13b5..078c0ab 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -184,7 +184,7 @@ def test_migration_nine_creates_capacity_ledger(self): version = store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 11) + self.assertEqual(version, 12) tables = { row[0] for row in store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_cli.py b/test/core/test_cli.py index d589c54..b0c4848 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -456,7 +456,7 @@ def build_python(): return candidate return None - def test_installed_wheel_contains_and_applies_migrations_through_eleven(self): + def test_installed_wheel_contains_and_applies_migrations_through_twelve(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") @@ -498,7 +498,7 @@ def test_installed_wheel_contains_and_applies_migrations_through_eleven(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 11 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 12 attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 14c7329..2630dd7 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 11, + 12, ) tables = { row[0] @@ -618,7 +618,7 @@ def build_python(): return candidate return None - def test_installed_wheel_applies_schema_four_to_eleven(self): + def test_installed_wheel_applies_schema_four_to_twelve(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel") @@ -684,7 +684,7 @@ def test_installed_wheel_applies_schema_four_to_eleven(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 11 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 12 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_learning.py b/test/core/test_learning.py index 0463752..601dd04 100644 --- a/test/core/test_learning.py +++ b/test/core/test_learning.py @@ -429,7 +429,7 @@ def test_current_schema_contains_outcome_ledger(self): store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0], - 11, + 12, ) columns = { row[1] for row in store.connection.execute("PRAGMA table_info(outcomes)") diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py new file mode 100644 index 0000000..16158a0 --- /dev/null +++ b/test/core/test_lifecycle.py @@ -0,0 +1,355 @@ +import copy +from datetime import datetime, timezone +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import threading +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.lifecycle import ( + guarded_change_violation, + profile_template_violation, + validate_profile_template, +) +from devsquad.store import ConflictError, Store + + +NOW = datetime(2026, 9, 29, 9, 0, tzinfo=timezone.utc) + + +def profile(profile_id, model_id, *, tools=None, pool="pool-a", billing="subscription"): + return { + "id": profile_id, + "harness": "fixture", + "model_family": "fixture-family", + "model_id": model_id, + "effort": {"value": "high", "transport": "native"}, + "required_tools": list(tools or ["read"]), + "permission_policy": "read_only", + "account_pool_id": pool, + "billing_mode": billing, + "quality_status": "proven", + "evidence_refs": ["tracked-fixture"], + } + + +def lifecycle_template(*, update_mode="guarded_auto"): + return { + "schema_version": 1, + "template_id": "template-review-deep-v1", + "alias": "review.deep", + "update_mode": update_mode, + "policy": {"id": "fixture-policy", "version": 3}, + "allowed_harnesses": ["fixture"], + "allowed_model_families": ["fixture-family"], + "allowed_account_pools": ["pool-a"], + "allowed_task_classes": ["fixture-review-small"], + "permission_policy": "read_only", + "allowed_tools": ["read", "web"], + "allowed_billing_modes": ["subscription"], + "gate": { + "min_evaluation_pairs": 1, + "min_held_out_pairs": 1, + "max_critical_defects": 0, + "max_latency_ratio": None, + "max_usage_ratio": None, + }, + } + + +def final_outcome(outcome_id, verdict): + return { + "schema_version": 1, + "outcome_id": outcome_id, + "kind": "final", + "verdict": verdict, + "selection_mode": "experimental", + "observed_at": NOW.isoformat(), + "corrects_outcome_id": None, + "summary": f"Lifecycle fixture {outcome_id} was {verdict}.", + "criteria": [], + "contributions": [], + "lead_repairs": [], + "evidence_refs": [f"{outcome_id}.json"], + } + + +class ProfileLifecycleTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-lifecycle-") + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], + check=True, + ) + (self.repo / "README").write_text("fixture\n") + subprocess.run(["git", "-C", str(self.repo), "add", "README"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.database = self.root / "state.sqlite3" + self.artifacts = self.root / "artifacts" + self.store = Store(self.database, self.artifacts) + self.incumbent = profile("profile-a", "model-a") + self.candidate = profile("profile-b", "model-b") + + def tearDown(self): + self.store.close() + self.temp.cleanup() + + def seed_experiment(self): + cases = [ + { + "case_id": "eval-1", "split": "evaluation", + "control_outcome_id": "control-eval", + "candidate_outcome_id": "candidate-eval", + }, + { + "case_id": "hold-1", "split": "held_out", + "control_outcome_id": "control-hold", + "candidate_outcome_id": "candidate-hold", + }, + ] + for case in cases: + for arm, verdict in (("control", "failed"), ("candidate", "succeeded")): + outcome_id = case[f"{arm}_outcome_id"] + claim = self.store.claim_start( + self.repo, f"run-{outcome_id}", {}, "owner", + ) + self.store.connection.execute( + "UPDATE runs SET state=?,phase=NULL WHERE id=?", + (verdict, claim.run_id), + ) + self.store.record_outcome( + claim.run_id, final_outcome(outcome_id, verdict), now=NOW, + ) + spec = { + "schema_version": 1, + "experiment_id": "experiment-profile-b", + "project_path": str(self.repo), + "question": "Should profile B replace profile A?", + "hypothesis": "Profile B improves held-out success.", + "evidence_availability": "tracked_fixture", + "variable": { + "kind": "profile_binding", + "alias": "review.deep", + "control_profile_id": "profile-a", + "candidate_profile_id": "profile-b", + }, + "cases": cases, + "gate": { + "min_evaluation_pairs": 1, + "min_held_out_pairs": 1, + "noninferiority_margin": 0.0, + "minimum_success_gain": 1.0, + "max_candidate_escaped_defects": 0, + }, + "budget": { + "max_cases": 2, + "max_worker_invocations": 0, + "wall_seconds": 60, + }, + "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, + } + return self.store.evaluate_learning_experiment(spec, now=NOW) + + def qualification(self, evaluation, *, qualification_id="qualification-b"): + return { + "schema_version": 1, + "qualification_id": qualification_id, + "alias": "review.deep", + "template_id": "template-review-deep-v1", + "candidate_profile": self.candidate, + "task_class": "fixture-review-small", + "experiment_id": evaluation["experiment"]["experiment_id"], + "evaluation_sha256": evaluation["evaluation_sha256"], + "source": { + "harness_version": "fixture 1.0", + "catalog_sha256": "a" * 64, + "model_revision": None, + }, + "budget": { + "max_cases": 2, + "used_cases": 2, + "max_worker_invocations": 0, + "worker_invocations": 0, + "max_wall_seconds": 60, + "wall_seconds": 5, + }, + "measured": { + "evaluation_pairs": 1, + "held_out_pairs": 1, + "critical_defects": 0, + "latency_ratio": None, + "usage_ratio": None, + }, + "verdict": "qualified", + "evidence_refs": ["experiment-profile-b", "evaluation.json"], + } + + @staticmethod + def promotion(decision_id, *, actor="human", expected=7): + return { + "schema_version": 1, + "decision_id": decision_id, + "action": "promote", + "alias": "review.deep", + "expected_binding_version": expected, + "qualification_id": "qualification-b", + "rollback_target": None, + "actor": actor, + "reason": "Held-out evidence passed the reviewed gate.", + "evidence_refs": ["experiment-profile-b", "evaluation.json"], + } + + def test_template_and_guarded_authority_boundaries_are_strict(self): + template = validate_profile_template(lifecycle_template()) + self.assertIsNone(profile_template_violation(self.candidate, template)) + paid = profile("paid", "model-paid", billing="paid_api") + self.assertEqual( + profile_template_violation(paid, template), + "billing_mode_not_allowed", + ) + expanded = profile("expanded", "model-expanded", tools=["read", "web"]) + self.assertEqual( + guarded_change_violation(self.incumbent, expanded), + "guarded_tool_expansion", + ) + invalid = lifecycle_template() + invalid["gate"]["min_held_out_pairs"] = 0 + with self.assertRaisesRegex(ContractError, "min_held_out_pairs"): + validate_profile_template(invalid) + + def test_qualification_cas_promotion_and_rollback_are_replay_safe(self): + template = lifecycle_template() + registered = self.store.register_profile_template(template, now=NOW) + replayed = self.store.register_profile_template(template, now=NOW) + self.assertFalse(registered["replayed"]) + self.assertTrue(replayed["replayed"]) + baseline = self.store.bootstrap_profile_binding( + template, self.incumbent, version=7, now=NOW, + ) + self.assertFalse(baseline["replayed"]) + self.assertTrue(self.store.bootstrap_profile_binding( + template, self.incumbent, version=7, now=NOW, + )["replayed"]) + evaluation = self.seed_experiment() + qualified = self.store.record_profile_qualification( + self.qualification(evaluation), now=NOW, + ) + self.assertEqual(qualified["gate_failures"], []) + self.assertTrue(self.store.record_profile_qualification( + self.qualification(evaluation), now=NOW, + )["replayed"]) + + barrier = threading.Barrier(2) + results = [] + + def promote(decision_id): + connection = Store(self.database, self.artifacts) + try: + barrier.wait(timeout=10) + results.append(connection.change_profile_binding( + self.promotion(decision_id), now=NOW, + )) + except Exception as exc: + results.append(exc) + finally: + connection.close() + + threads = [ + threading.Thread(target=promote, args=(decision_id,)) + for decision_id in ("decision-promote-a", "decision-promote-b") + ] + for thread in threads: + thread.start() + for thread in threads: + thread.join() + successes = [item for item in results if isinstance(item, dict)] + conflicts = [item for item in results if isinstance(item, ConflictError)] + self.assertEqual((len(successes), len(conflicts)), (1, 1), results) + receipt = successes[0]["receipt"] + self.assertEqual(receipt["to"]["binding_version"], 8) + self.assertTrue(receipt["affects_new_runs_only"]) + self.assertEqual(self.store.profile_binding("review.deep")["profile_id"], "profile-b") + winning_request = self.promotion(receipt["decision_id"]) + self.assertTrue(self.store.change_profile_binding( + winning_request, now=NOW, + )["replayed"]) + + rollback = { + "schema_version": 1, + "decision_id": "decision-rollback-a", + "action": "rollback", + "alias": "review.deep", + "expected_binding_version": 8, + "qualification_id": None, + "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, + "actor": "guarded_auto", + "reason": "Held-out regression requires the qualified predecessor.", + "evidence_refs": ["regression.json"], + } + rolled_back = self.store.change_profile_binding(rollback, now=NOW) + self.assertEqual(rolled_back["receipt"]["to"]["binding_version"], 9) + self.assertEqual(self.store.profile_binding("review.deep")["profile_id"], "profile-a") + self.assertEqual(len(self.store.profile_binding_decisions("review.deep")), 2) + + def test_insufficient_evidence_and_disabled_guarded_auto_cannot_promote(self): + reviewed = lifecycle_template(update_mode="reviewed") + self.store.bootstrap_profile_binding( + reviewed, self.incumbent, version=7, now=NOW, + ) + incomplete = self.qualification({ + "experiment": {"experiment_id": "unused"}, + "evaluation_sha256": "0" * 64, + }, qualification_id="qualification-incomplete") + incomplete.update({ + "experiment_id": None, + "evaluation_sha256": None, + "verdict": "incomplete", + "evidence_refs": [], + }) + saved = self.store.record_profile_qualification(incomplete, now=NOW) + self.assertIn("experiment_evidence_missing", saved["gate_failures"]) + with self.assertRaisesRegex(ContractError, "qualified candidate"): + self.store.change_profile_binding( + self.promotion("decision-insufficient"), now=NOW, + ) + + evaluation = self.seed_experiment() + self.store.record_profile_qualification( + self.qualification(evaluation), now=NOW, + ) + with self.assertRaisesRegex(ContractError, "not enabled"): + self.store.change_profile_binding( + self.promotion("decision-auto", actor="guarded_auto"), now=NOW, + ) + + def test_schema_twelve_contains_lifecycle_ledger(self): + version = self.store.connection.execute( + "SELECT MAX(version) FROM schema_migrations", + ).fetchone()[0] + self.assertEqual(version, 12) + tables = { + row[0] for row in self.store.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table'", + ) + } + self.assertTrue({ + "profile_templates", "concrete_profiles", "qualification_runs", + "profile_bindings", "profile_binding_versions", "binding_decisions", + } <= tables) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_store.py b/test/core/test_store.py index efb2335..c04301c 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -482,8 +482,8 @@ def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): ) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 11) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(12,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 12) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(13,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -498,7 +498,7 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 11) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 12) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") @@ -515,7 +515,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],11) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],12) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From 69f0c8e52e8342da2ea4fb4cddf35d35e004f51b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:07:28 -0700 Subject: [PATCH 106/197] feat: apply lifecycle bindings to new runs --- plugin/core/src/devsquad/service.py | 14 +++- plugin/core/src/devsquad/store.py | 85 ++++++++++++++++++++++ test/core/test_lifecycle.py | 105 ++++++++++++++++++++++++++++ 3 files changed, 202 insertions(+), 2 deletions(-) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 011bb0b..ef90487 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -462,16 +462,26 @@ def _resolve_snapshot( else: if capacity_store is None: raise ContractError("public preflight requires shared capacity state") + effective_registry = capacity_store.effective_profile_registry( + config_payloads["profiles_file"], + config_payloads["policy_file"], + ) snapshot["routing"] = load_routing( task, - config_payloads["profiles_file"], + effective_registry["profiles_payload"], config_payloads["policy_file"], availability=capacity_with_saved_observations( - config_payloads["profiles_file"], + effective_registry["profiles_payload"], config_payloads["policy_file"], capacity_store.capacity_snapshot, ), ) + snapshot["routing"]["profile_registry"].update({ + "source_sha256": effective_registry["source_sha256"], + "lifecycle_bindings": effective_registry[ + "lifecycle_bindings" + ], + }) if project_id is None or run_id is None: raise ContractError("public preflight requires run-owned workspace identity") if task["workflow"] == "branch-review": diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 0a00f79..32ff77e 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1840,6 +1840,91 @@ def profile_binding_decisions(self, alias: str) -> list[dict[str, Any]]: ) ] + def effective_profile_registry( + self, + profiles_payload: bytes | str, + policy_payload: bytes | str, + ) -> dict[str, Any]: + """Overlay policy-matched local bindings without editing project files.""" + from .router import _strict_json + from .validation import validate_policy, validate_profile_registry + + registry, source_sha256 = _strict_json( + profiles_payload, "profiles file", + ) + policy, policy_sha256 = _strict_json(policy_payload, "policy file") + validate_profile_registry(registry) + validate_policy(policy) + aliases = sorted({ + reference["id"] + for candidates in policy["roles"].values() + for reference in candidates + if reference["kind"] == "alias" + }) + if not aliases: + return { + "profiles_payload": profiles_payload, + "source_sha256": source_sha256, + "policy_sha256": policy_sha256, + "lifecycle_bindings": [], + } + placeholders = ",".join("?" for _ in aliases) + rows = self.connection.execute( + "SELECT b.alias,b.profile_id,b.version,b.qualification_id,b.updated_at," + "p.profile_json,p.profile_sha256,t.template_id,t.payload_sha256," + "t.policy_id,t.policy_version FROM profile_bindings b " + "JOIN concrete_profiles p ON p.profile_id=b.profile_id " + "JOIN profile_templates t ON t.template_id=b.template_id " + f"WHERE b.alias IN ({placeholders}) ORDER BY b.alias", + tuple(aliases), + ).fetchall() + applicable = [ + row for row in rows + if (row["policy_id"] == policy["id"] + and row["policy_version"] == policy["version"]) + ] + if not applicable: + return { + "profiles_payload": profiles_payload, + "source_sha256": source_sha256, + "policy_sha256": policy_sha256, + "lifecycle_bindings": [], + } + effective = json.loads(canonical_json(registry)) + profiles = {profile["id"]: profile for profile in effective["profiles"]} + lifecycle_bindings = [] + for row in applicable: + profile = json.loads(row["profile_json"]) + existing = profiles.get(profile["id"]) + if existing is not None and canonical_json(existing) != canonical_json(profile): + raise ConflictError( + "runtime profile conflicts with the project profile registry", + ) + if existing is None: + effective["profiles"].append(profile) + profiles[profile["id"]] = profile + effective["bindings"][row["alias"]] = { + "profile_id": row["profile_id"], + "version": row["version"], + } + lifecycle_bindings.append({ + "alias": row["alias"], + "profile_id": row["profile_id"], + "profile_sha256": row["profile_sha256"], + "version": row["version"], + "template_id": row["template_id"], + "template_sha256": row["payload_sha256"], + "qualification_id": row["qualification_id"], + "updated_at": row["updated_at"], + }) + effective["profiles"].sort(key=lambda profile: profile["id"]) + return { + "profiles_payload": (canonical_json(effective) + "\n").encode(), + "source_sha256": source_sha256, + "policy_sha256": policy_sha256, + "lifecycle_bindings": lifecycle_bindings, + } + def worker_invocations(self, run_id: str) -> int: """Count attempts whose durable runner actually crossed the launch fence.""" return self.connection.execute( diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index 16158a0..22fdc42 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -17,6 +17,7 @@ profile_template_violation, validate_profile_template, ) +from devsquad.router import load_routing from devsquad.store import ConflictError, Store @@ -80,6 +81,59 @@ def final_outcome(outcome_id, verdict): } +def routing_policy(): + return { + "schema_version": 1, + "id": "fixture-policy", + "version": 3, + "roles": {"reviewer": [{"kind": "alias", "id": "review.deep"}]}, + "task_classes": {"fixture-review-small": "proven"}, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": False, + "account_pools": { + "pool-a": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 2, + "unknown_capacity_policy": "allow_bounded", + }, + }, + "experiment_budget": {}, + } + + +def review_task(repo, *, pinned_profile_id=None): + routing = { + "profiles_file": "profiles.json", + "policy_file": "policy.json", + } + if pinned_profile_id is not None: + routing["overrides"] = { + "reviewer": {"profile_id": pinned_profile_id, "fallback": "none"}, + } + return { + "schema_version": 1, + "project": { + "repo_path": str(repo), "base_ref": "HEAD", "target_ref": "HEAD", + }, + "workflow": "branch-review", + "goal": "Verify lifecycle routing.", + "task_class": "fixture-review-small", + "acceptance": [{ + "id": "routing", "description": "The expected profile is frozen.", + "evidence_kind": "review", + }], + "checks": [], + "scope": {"read_paths": ["README"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": routing, + "budget": { + "wall_seconds": 60, "max_worker_invocations": 1, + "max_revisions": 0, "max_fallbacks_per_step": 0, + }, + "origin": {"surface": "test"}, + } + + class ProfileLifecycleTest(unittest.TestCase): def setUp(self): self.temp = tempfile.TemporaryDirectory(prefix="devsquad-lifecycle-") @@ -243,6 +297,25 @@ def test_qualification_cas_promotion_and_rollback_are_replay_safe(self): self.assertTrue(self.store.bootstrap_profile_binding( template, self.incumbent, version=7, now=NOW, )["replayed"]) + registry = { + "schema_version": 1, + "profiles": [self.incumbent], + "bindings": { + "review.deep": {"profile_id": "profile-a", "version": 7}, + }, + } + profiles_payload = json.dumps(registry, sort_keys=True) + "\n" + policy_payload = json.dumps(routing_policy(), sort_keys=True) + "\n" + frozen_input = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + frozen_routing = load_routing( + review_task(self.repo), frozen_input["profiles_payload"], policy_payload, + ) + self.assertEqual( + frozen_routing["roles"]["reviewer"]["selected"]["profile_id"], + "profile-a", + ) evaluation = self.seed_experiment() qualified = self.store.record_profile_qualification( self.qualification(evaluation), now=NOW, @@ -282,6 +355,28 @@ def promote(decision_id): self.assertEqual(receipt["to"]["binding_version"], 8) self.assertTrue(receipt["affects_new_runs_only"]) self.assertEqual(self.store.profile_binding("review.deep")["profile_id"], "profile-b") + promoted_input = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + promoted_routing = load_routing( + review_task(self.repo), promoted_input["profiles_payload"], policy_payload, + ) + self.assertEqual( + promoted_routing["roles"]["reviewer"]["selected"]["profile_id"], + "profile-b", + ) + self.assertEqual( + frozen_routing["roles"]["reviewer"]["selected"]["profile_id"], + "profile-a", + ) + pinned = load_routing( + review_task(self.repo, pinned_profile_id="profile-a"), + promoted_input["profiles_payload"], policy_payload, + ) + self.assertEqual( + pinned["roles"]["reviewer"]["selected"]["profile_id"], + "profile-a", + ) winning_request = self.promotion(receipt["decision_id"]) self.assertTrue(self.store.change_profile_binding( winning_request, now=NOW, @@ -303,6 +398,16 @@ def promote(decision_id): self.assertEqual(rolled_back["receipt"]["to"]["binding_version"], 9) self.assertEqual(self.store.profile_binding("review.deep")["profile_id"], "profile-a") self.assertEqual(len(self.store.profile_binding_decisions("review.deep")), 2) + rollback_input = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + rollback_routing = load_routing( + review_task(self.repo), rollback_input["profiles_payload"], policy_payload, + ) + self.assertEqual( + rollback_routing["roles"]["reviewer"]["selected"]["binding"]["version"], + 9, + ) def test_insufficient_evidence_and_disabled_guarded_auto_cannot_promote(self): reviewed = lifecycle_template(update_mode="reviewed") From d7fde77e69925461b13b05510845effb10503a16 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:11:03 -0700 Subject: [PATCH 107/197] feat: expose profile lifecycle operations --- plugin/core/src/devsquad/cli.py | 46 ++++++++++++++++++ plugin/core/src/devsquad/lifecycle.py | 46 ++++++++++++++++++ plugin/core/src/devsquad/service.py | 67 +++++++++++++++++++++++++++ test/core/test_cli.py | 45 ++++++++++++++++++ test/core/test_lifecycle.py | 13 ++++++ 5 files changed, 217 insertions(+) diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 8859f36..7941def 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -182,6 +182,34 @@ def command_learn_propose(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).learning_propose(args.project)), 0 +def command_profile_template_add(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_template_add( + _read_json(args.file, "profile template file"), + )), 0 + + +def command_profile_binding_bootstrap(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_binding_bootstrap( + _read_json(args.file, "profile binding bootstrap file"), + )), 0 + + +def command_profile_qualification_add(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_qualification_add( + _read_json(args.file, "profile qualification file"), + )), 0 + + +def command_profile_binding_change(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_binding_change( + _read_json(args.file, "profile binding change file"), + )), 0 + + +def command_profile_binding_show(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_binding_status(args.alias)), 0 + + def command_status(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).status(args.run)), 0 def command_events(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).events(args.run, args.after, args.limit)), 0 def command_result(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).result(args.run)), 0 @@ -305,6 +333,24 @@ def parser() -> argparse.ArgumentParser: propose.add_argument("--json", action="store_true") propose.add_argument("--runtime-dir", default=runtime_default) propose.set_defaults(func=command_learn_propose) + profile = sub.add_parser("profile") + profile_sub = profile.add_subparsers(dest="profile_command", required=True) + for name, fn in ( + ("template-add", command_profile_template_add), + ("binding-bootstrap", command_profile_binding_bootstrap), + ("qualification-add", command_profile_qualification_add), + ("binding-change", command_profile_binding_change), + ): + operation = profile_sub.add_parser(name) + operation.add_argument("--file", required=True) + operation.add_argument("--json", action="store_true") + operation.add_argument("--runtime-dir", default=runtime_default) + operation.set_defaults(func=fn) + binding_show = profile_sub.add_parser("binding-show") + binding_show.add_argument("alias") + binding_show.add_argument("--json", action="store_true") + binding_show.add_argument("--runtime-dir", default=runtime_default) + binding_show.set_defaults(func=command_profile_binding_show) handoff = sub.add_parser("handoff") handoff_sub = handoff.add_subparsers(dest="handoff_command", required=True) claim = handoff_sub.add_parser("claim") diff --git a/plugin/core/src/devsquad/lifecycle.py b/plugin/core/src/devsquad/lifecycle.py index 7618dfe..e077d61 100644 --- a/plugin/core/src/devsquad/lifecycle.py +++ b/plugin/core/src/devsquad/lifecycle.py @@ -337,3 +337,49 @@ def validate_binding_change(value: dict[str, Any]) -> dict[str, Any]: def validate_recorded_at(value: Any) -> str: return _timestamp(value, "recorded_at") + + +def render_binding_decision_markdown(receipt: dict[str, Any]) -> str: + """Render an inspectable local promotion/rollback decision receipt.""" + required = { + "schema_version", "decision_id", "action", "alias", "actor", "reason", + "evidence_refs", "from", "to", "qualification_id", "qualification", + "policy", "rollback_target", "requested_rollback_target", "effective_at", + "affects_new_runs_only", + } + if not isinstance(receipt, dict) or set(receipt) != required: + raise ContractError("binding decision receipt is invalid") + if (receipt["schema_version"] != 1 + or receipt["action"] not in {"promote", "rollback"} + or receipt["actor"] not in {"human", "guarded_auto"} + or receipt["affects_new_runs_only"] is not True): + raise ContractError("binding decision receipt values are invalid") + _timestamp(receipt["effective_at"], "effective_at") + lines = [ + f"# Profile binding decision {receipt['decision_id']}", + "", + f"- Action: `{receipt['action']}`", + f"- Alias: `{receipt['alias']}`", + f"- Actor: `{receipt['actor']}`", + f"- Effective: `{receipt['effective_at']}`", + "- Scope: new runs only", + f"- Reason: {receipt['reason']}", + "", + "## Binding change", + "", + f"- From: `{receipt['from']['profile_id']}` at version " + f"{receipt['from']['binding_version']}", + f"- To: `{receipt['to']['profile_id']}` at version " + f"{receipt['to']['binding_version']}", + f"- Rollback: `{receipt['rollback_target']['profile_id']}` at version " + f"{receipt['rollback_target']['binding_version']}", + f"- Policy: `{receipt['policy']['id']}` version {receipt['policy']['version']}", + "", + "## Evidence", + "", + ] + lines.extend( + [f"- `{reference}`" for reference in receipt["evidence_refs"]] + or ["- None."] + ) + return "\n".join(lines) + "\n" diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index ef90487..8b73475 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -892,6 +892,73 @@ def learning_propose(self, project: str | Path) -> dict[str, Any]: }, } + def profile_template_add(self, template: dict[str, Any]) -> dict[str, Any]: + store = self._store() + try: + return store.register_profile_template(template) + finally: + store.close() + + def profile_binding_bootstrap(self, request: dict[str, Any]) -> dict[str, Any]: + if not isinstance(request, dict) or set(request) != { + "template", "profile", "version", + }: + raise ContractError("profile binding bootstrap fields are invalid") + store = self._store() + try: + return store.bootstrap_profile_binding( + request["template"], request["profile"], version=request["version"], + ) + finally: + store.close() + + def profile_qualification_add( + self, qualification: dict[str, Any], + ) -> dict[str, Any]: + store = self._store() + try: + return store.record_profile_qualification(qualification) + finally: + store.close() + + def profile_binding_change(self, change: dict[str, Any]) -> dict[str, Any]: + from .lifecycle import render_binding_decision_markdown + + store = self._store() + try: + result = store.change_profile_binding(change) + finally: + store.close() + receipt = result["receipt"] + json_content = (canonical_json(receipt) + "\n").encode() + markdown_content = render_binding_decision_markdown(receipt).encode() + directory = self.runtime / "learning" / "decisions" + directory.mkdir(parents=True, exist_ok=True) + return { + **result, + "artifacts": { + "json": self._finalize_learning_file( + directory, f"{receipt['decision_id']}.json", json_content, + ), + "markdown": self._finalize_learning_file( + directory, f"{receipt['decision_id']}.md", markdown_content, + ), + }, + } + + def profile_binding_status(self, alias: str) -> dict[str, Any]: + store = self._store() + try: + binding = store.profile_binding(alias) + if binding is None: + raise ContractError("profile binding does not exist") + return { + "binding": binding, + "decisions": store.profile_binding_decisions(alias), + } + finally: + store.close() + @staticmethod def _status_capacity(store: Store, run: dict[str, Any]) -> dict[str, Any] | None: try: diff --git a/test/core/test_cli.py b/test/core/test_cli.py index b0c4848..cf56c1f 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -194,6 +194,51 @@ def test_learn_propose_dispatches_project(self): self.assert_success_envelope(payload, response) service.learning_propose.assert_called_once_with(str(self.root)) + def test_profile_lifecycle_commands_dispatch_strict_files(self): + operations = [ + ( + "template-add", "profile_template_add", "profile-template.json", + {"schema_version": 1, "template_id": "template-1"}, + ), + ( + "binding-bootstrap", "profile_binding_bootstrap", "bootstrap.json", + {"template": {}, "profile": {}, "version": 1}, + ), + ( + "qualification-add", "profile_qualification_add", "qualification.json", + {"schema_version": 1, "qualification_id": "qualification-1"}, + ), + ( + "binding-change", "profile_binding_change", "change.json", + {"schema_version": 1, "decision_id": "decision-1"}, + ), + ] + for command, method_name, filename, document in operations: + with self.subTest(command=command): + path = self.root / filename + path.write_text(json.dumps(document)) + service = mock.Mock() + response = {"operation": command} + getattr(service, method_name).return_value = response + code, payload, stderr = self.invoke([ + "profile", command, "--file", str(path), + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + getattr(service, method_name).assert_called_once_with(document) + + service = mock.Mock() + response = {"binding": {"alias": "review.deep"}, "decisions": []} + service.profile_binding_status.return_value = response + code, payload, stderr = self.invoke([ + "profile", "binding-show", "review.deep", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.profile_binding_status.assert_called_once_with("review.deep") + def test_handoff_claim_renew_and_complete_dispatch_parsed_objects(self): claim_payload = { "schema_version": 1, diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index 22fdc42..4a94593 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -1,5 +1,6 @@ import copy from datetime import datetime, timezone +import hashlib import json from pathlib import Path import subprocess @@ -18,6 +19,7 @@ validate_profile_template, ) from devsquad.router import load_routing +from devsquad.service import Service from devsquad.store import ConflictError, Store @@ -408,6 +410,17 @@ def promote(decision_id): rollback_routing["roles"]["reviewer"]["selected"]["binding"]["version"], 9, ) + service = Service(self.root) + service_change = service.profile_binding_change( + self.promotion("decision-service", expected=9), + ) + self.assertEqual(service_change["receipt"]["to"]["binding_version"], 10) + for artifact in service_change["artifacts"].values(): + content = Path(artifact["path"]).read_bytes() + self.assertEqual(hashlib.sha256(content).hexdigest(), artifact["sha256"]) + status = service.profile_binding_status("review.deep") + self.assertEqual(status["binding"]["profile_id"], "profile-b") + self.assertEqual(len(status["decisions"]), 3) def test_insufficient_evidence_and_disabled_guarded_auto_cannot_promote(self): reviewed = lifecycle_template(update_mode="reviewed") From 4b27e0c5f70cfd6ce77c17748fd81164b6ee1dcd Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:14:45 -0700 Subject: [PATCH 108/197] feat: bind rollback to regression evidence --- plugin/core/src/devsquad/lifecycle.py | 13 +++++++--- plugin/core/src/devsquad/store.py | 31 +++++++++++++++++++++++ test/core/test_lifecycle.py | 36 +++++++++++++++++++++------ 3 files changed, 68 insertions(+), 12 deletions(-) diff --git a/plugin/core/src/devsquad/lifecycle.py b/plugin/core/src/devsquad/lifecycle.py index e077d61..ccbd9ef 100644 --- a/plugin/core/src/devsquad/lifecycle.py +++ b/plugin/core/src/devsquad/lifecycle.py @@ -26,7 +26,7 @@ BINDING_CHANGE_FIELDS = { "schema_version", "decision_id", "action", "alias", "expected_binding_version", "qualification_id", "rollback_target", - "actor", "reason", "evidence_refs", + "experiment_id", "evaluation_sha256", "actor", "reason", "evidence_refs", } SHA256_LENGTH = 64 @@ -309,8 +309,11 @@ def validate_binding_change(value: dict[str, Any]) -> dict[str, Any]: rollback_target = value["rollback_target"] if value["action"] == "promote": _identifier(qualification_id, "binding change qualification_id") - if rollback_target is not None: - raise ContractError("promotion cannot specify a rollback target") + if (rollback_target is not None or value["experiment_id"] is not None + or value["evaluation_sha256"] is not None): + raise ContractError( + "promotion cannot specify rollback experiment evidence", + ) else: if qualification_id is not None: raise ContractError("rollback cannot specify a qualification") @@ -318,6 +321,8 @@ def validate_binding_change(value: dict[str, Any]) -> dict[str, Any]: _identifier(rollback_target["profile_id"], "rollback profile_id") if type(rollback_target["binding_version"]) is not int or rollback_target["binding_version"] < 1: raise ContractError("rollback binding_version is invalid") + _identifier(value["experiment_id"], "rollback experiment_id") + _sha256(value["evaluation_sha256"], "rollback evaluation_sha256") if value["actor"] not in {"human", "guarded_auto"}: raise ContractError("binding change actor is invalid") reason = _identifier(value["reason"], "binding change reason") @@ -345,7 +350,7 @@ def render_binding_decision_markdown(receipt: dict[str, Any]) -> str: "schema_version", "decision_id", "action", "alias", "actor", "reason", "evidence_refs", "from", "to", "qualification_id", "qualification", "policy", "rollback_target", "requested_rollback_target", "effective_at", - "affects_new_runs_only", + "rollback_evaluation", "affects_new_runs_only", } if not isinstance(receipt, dict) or set(receipt) != required: raise ContractError("binding decision receipt is invalid") diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 32ff77e..0c68a98 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1667,6 +1667,7 @@ def change_profile_binding( "guarded automatic promotion is not enabled", ) qualification_payload = None + rollback_evaluation = None if request["action"] == "promote": qualification = self.connection.execute( "SELECT q.alias,q.template_id,q.profile_id,q.verdict," @@ -1738,6 +1739,35 @@ def change_profile_binding( raise ContractError( "guarded automation cannot change lifecycle policy", ) + regression = self.connection.execute( + "SELECT spec_json,evaluation_json,evaluation_sha256,verdict " + "FROM experiments WHERE experiment_id=?", + (request["experiment_id"],), + ).fetchone() + if (regression is None + or regression["evaluation_sha256"] + != request["evaluation_sha256"] + or regression["verdict"] != "no_change"): + raise ContractError( + "rollback requires saved no-change regression evidence", + ) + regression_spec = json.loads(regression["spec_json"]) + regression_evaluation = json.loads(regression["evaluation_json"]) + variable = regression_spec["variable"] + if (variable["alias"] != request["alias"] + or variable["candidate_profile_id"] != current["profile_id"] + or variable["control_profile_id"] != target_profile["id"]): + raise ContractError( + "rollback experiment does not compare the active and target profiles", + ) + rollback_evaluation = { + "experiment_id": request["experiment_id"], + "evaluation_sha256": request["evaluation_sha256"], + "verdict": regression["verdict"], + "reasons": regression_evaluation["reasons"], + "metrics": regression_evaluation["metrics"], + "failures": regression_evaluation["failures"], + } if target_profile["id"] == current["profile_id"]: raise ConflictError("binding already targets the requested profile") new_version = current["version"] + 1 @@ -1774,6 +1804,7 @@ def change_profile_binding( request["rollback_target"] if request["action"] == "rollback" else None ), + "rollback_evaluation": rollback_evaluation, "effective_at": recorded_at, "affects_new_runs_only": True, } diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index 4a94593..d42d5b9 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -163,21 +163,32 @@ def tearDown(self): self.store.close() self.temp.cleanup() - def seed_experiment(self): + def seed_experiment( + self, + *, + experiment_id="experiment-profile-b", + candidate_succeeds=True, + ): + suffix = experiment_id.replace("experiment-", "") cases = [ { "case_id": "eval-1", "split": "evaluation", - "control_outcome_id": "control-eval", - "candidate_outcome_id": "candidate-eval", + "control_outcome_id": f"control-eval-{suffix}", + "candidate_outcome_id": f"candidate-eval-{suffix}", }, { "case_id": "hold-1", "split": "held_out", - "control_outcome_id": "control-hold", - "candidate_outcome_id": "candidate-hold", + "control_outcome_id": f"control-hold-{suffix}", + "candidate_outcome_id": f"candidate-hold-{suffix}", }, ] for case in cases: - for arm, verdict in (("control", "failed"), ("candidate", "succeeded")): + verdicts = ( + (("control", "failed"), ("candidate", "succeeded")) + if candidate_succeeds + else (("control", "succeeded"), ("candidate", "failed")) + ) + for arm, verdict in verdicts: outcome_id = case[f"{arm}_outcome_id"] claim = self.store.claim_start( self.repo, f"run-{outcome_id}", {}, "owner", @@ -191,7 +202,7 @@ def seed_experiment(self): ) spec = { "schema_version": 1, - "experiment_id": "experiment-profile-b", + "experiment_id": experiment_id, "project_path": str(self.repo), "question": "Should profile B replace profile A?", "hypothesis": "Profile B improves held-out success.", @@ -263,6 +274,8 @@ def promotion(decision_id, *, actor="human", expected=7): "expected_binding_version": expected, "qualification_id": "qualification-b", "rollback_target": None, + "experiment_id": None, + "evaluation_sha256": None, "actor": actor, "reason": "Held-out evidence passed the reviewed gate.", "evidence_refs": ["experiment-profile-b", "evaluation.json"], @@ -384,6 +397,11 @@ def promote(decision_id): winning_request, now=NOW, )["replayed"]) + regression = self.seed_experiment( + experiment_id="experiment-profile-b-regression", + candidate_succeeds=False, + ) + self.assertEqual(regression["evaluation"]["verdict"], "no_change") rollback = { "schema_version": 1, "decision_id": "decision-rollback-a", @@ -392,9 +410,11 @@ def promote(decision_id): "expected_binding_version": 8, "qualification_id": None, "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, + "experiment_id": "experiment-profile-b-regression", + "evaluation_sha256": regression["evaluation_sha256"], "actor": "guarded_auto", "reason": "Held-out regression requires the qualified predecessor.", - "evidence_refs": ["regression.json"], + "evidence_refs": ["experiment-profile-b-regression", "regression.json"], } rolled_back = self.store.change_profile_binding(rollback, now=NOW) self.assertEqual(rolled_back["receipt"]["to"]["binding_version"], 9) From 398ae6acd6316db7f87654d3815b0115e6e9dffb Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:23:58 -0700 Subject: [PATCH 109/197] feat: record targeted catalog drift --- plugin/core/src/devsquad/catalog.py | 94 ++++++++++++++++++++++++++++- test/core/test_m1.py | 94 +++++++++++++++++++++++++++++ 2 files changed, 187 insertions(+), 1 deletion(-) diff --git a/plugin/core/src/devsquad/catalog.py b/plugin/core/src/devsquad/catalog.py index 5e8f38a..44999ea 100644 --- a/plugin/core/src/devsquad/catalog.py +++ b/plugin/core/src/devsquad/catalog.py @@ -10,6 +10,8 @@ from typing import Any, Iterable from .contracts import ContractError +from .store import canonical_json +from .validation import validate_profile def model_fingerprint(harness: str, version: str | None, model: dict[str, Any]) -> str: @@ -19,12 +21,20 @@ def model_fingerprint(harness: str, version: str | None, model: dict[str, Any]) def normalize_models(harness: str, version: str | None, models: Iterable[dict[str, Any]]) -> list[dict[str, Any]]: normalized = [] + identifiers = set() for raw in models: + if not isinstance(raw, dict): + raise ContractError("catalog model must be an object") model_id = raw.get("id") or raw.get("model") if not isinstance(model_id, str) or not model_id: raise ContractError("catalog model missing id") + if model_id in identifiers: + raise ContractError("catalog model ids must be unique") + identifiers.add(model_id) effort_values = raw.get("supportedReasoningEfforts") or raw.get("supported_reasoning_efforts") or [] efforts = [item.get("reasoningEffort") if isinstance(item, dict) else item for item in effort_values] + if not all(isinstance(effort, str) and effort for effort in efforts): + raise ContractError("catalog model effort metadata is invalid") normalized.append({ "id": model_id, "display_name": raw.get("displayName") or raw.get("display_name") or model_id, @@ -32,6 +42,7 @@ def normalize_models(harness: str, version: str | None, models: Iterable[dict[st "default_effort": raw.get("defaultReasoningEffort") or raw.get("default_reasoning_effort"), "supported_efforts": efforts, "modalities": raw.get("inputModalities") or raw.get("input_modalities") or [], + "revision": raw.get("modelRevision") or raw.get("revision"), "is_default": bool(raw.get("isDefault") or raw.get("is_default")), "qualification": "unqualified", "fingerprint": model_fingerprint(harness, version, raw), @@ -39,7 +50,85 @@ def normalize_models(harness: str, version: str | None, models: Iterable[dict[st return normalized -def update_last_good(path: Path, *, harness: str, version: str | None, models: Iterable[dict[str, Any]] | None, complete: bool, error: str | None = None) -> dict[str, Any]: +def analyze_catalog_drift( + previous: dict[str, Any] | None, + current: dict[str, Any], + *, + profiles: Iterable[dict[str, Any]] | None = None, +) -> dict[str, Any]: + """Identify exact model/profile drift without changing any binding.""" + if (not isinstance(current, dict) or current.get("complete") is not True + or not isinstance(current.get("harness"), str) + or not isinstance(current.get("models"), list)): + raise ContractError("current catalog snapshot is invalid") + if previous is not None and ( + not isinstance(previous, dict) or previous.get("complete") is not True + or previous.get("harness") != current["harness"] + or not isinstance(previous.get("models"), list)): + raise ContractError("previous catalog snapshot is invalid") + + def models_by_id(snapshot): + result = {} + for model in snapshot.get("models", []): + if (not isinstance(model, dict) + or not isinstance(model.get("id"), str) + or model["id"] in result): + raise ContractError("catalog snapshot model identity is invalid") + result[model["id"]] = model + return result + + old_models = {} if previous is None else models_by_id(previous) + new_models = models_by_id(current) + added = sorted(set(new_models) - set(old_models)) + removed = sorted(set(old_models) - set(new_models)) + changed = sorted( + model_id for model_id in set(old_models) & set(new_models) + if old_models[model_id].get("fingerprint") + != new_models[model_id].get("fingerprint") + ) + unknown_revision = sorted( + model_id for model_id in changed + if (old_models[model_id].get("revision") is None + or new_models[model_id].get("revision") is None) + ) + profile_rows = [] if profiles is None else list(profiles) + for profile in profile_rows: + validate_profile(profile) + affected_ids = set(changed) | set(removed) + affected_profiles = sorted( + profile["id"] for profile in profile_rows + if (profile["harness"] == current["harness"] + and profile["model_id"] in affected_ids) + ) + unavailable_profiles = sorted( + profile["id"] for profile in profile_rows + if (profile["harness"] == current["harness"] + and profile["model_id"] in removed) + ) + previous_sha256 = ( + hashlib.sha256(canonical_json(previous).encode()).hexdigest() + if previous is not None else None + ) + return { + "schema_version": 1, + "harness": current["harness"], + "previous_sha256": previous_sha256, + "current_sha256": hashlib.sha256( + canonical_json(current).encode(), + ).hexdigest(), + "added_model_ids": added, + "removed_model_ids": removed, + "changed_model_ids": changed, + "same_id_revision_unknown": unknown_revision, + "affected_profile_ids": affected_profiles, + "unavailable_profile_ids": unavailable_profiles, + "profile_scope": "provided" if profiles is not None else "unavailable", + "unqualified_candidate_ids": added, + "binding_changes_applied": False, + } + + +def update_last_good(path: Path, *, harness: str, version: str | None, models: Iterable[dict[str, Any]] | None, complete: bool, error: str | None = None, profiles: Iterable[dict[str, Any]] | None = None) -> dict[str, Any]: old = json.loads(path.read_text()) if path.exists() else None now = datetime.now(timezone.utc).isoformat() if error or not complete or models is None: @@ -49,6 +138,9 @@ def update_last_good(path: Path, *, harness: str, version: str | None, models: I return old raise ContractError(error or "catalog response incomplete and no last-good snapshot exists") value = {"schema_version": 1, "harness": harness, "harness_version": version, "fetched_at": now, "complete": True, "models": normalize_models(harness, version, models), "last_refresh": {"at": now, "status": "ok", "error": None}} + value["catalog_change"] = analyze_catalog_drift( + old, value, profiles=profiles, + ) _atomic_json(path, value) return value diff --git a/test/core/test_m1.py b/test/core/test_m1.py index 070d4c2..c1bfdc3 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -256,6 +256,22 @@ def test_complete_pagination_empty_page_and_repeated_cursor(self): class CatalogTest(unittest.TestCase): + @staticmethod + def profile(profile_id, model_id, *, harness="codex"): + return { + "id": profile_id, + "harness": harness, + "model_family": "gpt", + "model_id": model_id, + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": f"{harness}-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["catalog-fixture"], + } + def test_incomplete_refresh_retains_last_good(self): with tempfile.TemporaryDirectory() as tmp: path = Path(tmp) / "catalog.json" @@ -270,6 +286,84 @@ def test_discovered_model_is_unqualified_and_unknown_family_stays_unknown(self): self.assertEqual(value["models"][0]["qualification"], "unqualified") self.assertIsNone(value["models"][0]["family"]) + def test_changed_effort_revalidates_only_affected_profiles(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "catalog.json" + profiles = [ + self.profile("profile-a", "gpt-a"), + self.profile("profile-b", "gpt-b"), + self.profile("profile-other", "gpt-a", harness="other"), + ] + update_last_good( + path, harness="codex", version="1", complete=True, + models=[ + {"id": "gpt-a", "modelRevision": "a1", "supportedReasoningEfforts": ["low"]}, + {"id": "gpt-b", "modelRevision": "b1", "supportedReasoningEfforts": ["low"]}, + ], + profiles=profiles, + ) + changed = update_last_good( + path, harness="codex", version="1", complete=True, + models=[ + {"id": "gpt-a", "modelRevision": "a2", "supportedReasoningEfforts": ["low", "high"]}, + {"id": "gpt-b", "modelRevision": "b1", "supportedReasoningEfforts": ["low"]}, + ], + profiles=profiles, + )["catalog_change"] + self.assertEqual(changed["changed_model_ids"], ["gpt-a"]) + self.assertEqual(changed["affected_profile_ids"], ["profile-a"]) + self.assertEqual(changed["same_id_revision_unknown"], []) + self.assertEqual(changed["unavailable_profile_ids"], []) + self.assertFalse(changed["binding_changes_applied"]) + + def test_same_id_unknown_revision_added_and_removed_models_stay_safe(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "catalog.json" + profiles = [ + self.profile("profile-a", "gpt-a"), + self.profile("profile-b", "gpt-b"), + ] + update_last_good( + path, harness="codex", version="1", complete=True, + models=[ + {"id": "gpt-a", "supportedReasoningEfforts": ["low"]}, + {"id": "gpt-b", "supportedReasoningEfforts": ["low"]}, + ], + profiles=profiles, + ) + changed = update_last_good( + path, harness="codex", version="1", complete=True, + models=[ + {"id": "gpt-a", "supportedReasoningEfforts": ["low", "high"]}, + {"id": "gpt-new", "supportedReasoningEfforts": ["low"]}, + ], + profiles=profiles, + ) + drift = changed["catalog_change"] + self.assertEqual(drift["changed_model_ids"], ["gpt-a"]) + self.assertEqual(drift["same_id_revision_unknown"], ["gpt-a"]) + self.assertEqual(drift["added_model_ids"], ["gpt-new"]) + self.assertEqual(drift["unqualified_candidate_ids"], ["gpt-new"]) + self.assertEqual(drift["removed_model_ids"], ["gpt-b"]) + self.assertEqual(drift["affected_profile_ids"], ["profile-a", "profile-b"]) + self.assertEqual(drift["unavailable_profile_ids"], ["profile-b"]) + self.assertEqual( + next(model for model in changed["models"] if model["id"] == "gpt-new")["qualification"], + "unqualified", + ) + self.assertFalse(drift["binding_changes_applied"]) + + def test_duplicate_catalog_model_ids_are_rejected(self): + with tempfile.TemporaryDirectory() as tmp: + with self.assertRaisesRegex(ContractError, "unique"): + update_last_good( + Path(tmp) / "catalog.json", + harness="codex", + version="1", + complete=True, + models=[{"id": "gpt-a"}, {"model": "gpt-a"}], + ) + if __name__ == "__main__": unittest.main() From ca4ee73040b8c8ed36d397e65a2c420fd22ef523 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:30:41 -0700 Subject: [PATCH 110/197] feat: fall back from unavailable profile bindings --- plugin/core/src/devsquad/catalog.py | 57 ++++++- plugin/core/src/devsquad/cli.py | 7 + plugin/core/src/devsquad/lifecycle.py | 40 +++++ plugin/core/src/devsquad/service.py | 18 ++- plugin/core/src/devsquad/store.py | 205 ++++++++++++++++++++++++++ test/core/test_cli.py | 4 + test/core/test_lifecycle.py | 121 +++++++++++++++ 7 files changed, 448 insertions(+), 4 deletions(-) diff --git a/plugin/core/src/devsquad/catalog.py b/plugin/core/src/devsquad/catalog.py index 44999ea..f6977c0 100644 --- a/plugin/core/src/devsquad/catalog.py +++ b/plugin/core/src/devsquad/catalog.py @@ -14,6 +14,59 @@ from .validation import validate_profile +CATALOG_CHANGE_FIELDS = { + "schema_version", "harness", "previous_sha256", "current_sha256", + "added_model_ids", "removed_model_ids", "changed_model_ids", + "same_id_revision_unknown", "affected_profile_ids", + "unavailable_profile_ids", "profile_scope", "unqualified_candidate_ids", + "binding_changes_applied", +} + + +def validate_catalog_change(value: dict[str, Any]) -> dict[str, Any]: + """Validate complete-catalog drift evidence before lifecycle mutation.""" + if not isinstance(value, dict) or set(value) != CATALOG_CHANGE_FIELDS: + raise ContractError("catalog change fields are invalid") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("catalog change schema_version is invalid") + if not isinstance(value["harness"], str) or not value["harness"]: + raise ContractError("catalog change harness is invalid") + for field in ("previous_sha256", "current_sha256"): + digest = value[field] + if (digest is None and field == "previous_sha256"): + continue + if (not isinstance(digest, str) or len(digest) != 64 + or any(character not in "0123456789abcdef" for character in digest)): + raise ContractError(f"catalog change {field} is invalid") + list_fields = CATALOG_CHANGE_FIELDS - { + "schema_version", "harness", "previous_sha256", "current_sha256", + "profile_scope", "binding_changes_applied", + } + normalized = dict(value) + for field in list_fields: + items = value[field] + if (not isinstance(items, list) + or any(not isinstance(item, str) or not item for item in items) + or len(items) != len(set(items))): + raise ContractError(f"catalog change {field} is invalid") + normalized[field] = sorted(items) + if value["profile_scope"] not in {"provided", "unavailable"}: + raise ContractError("catalog change profile_scope is invalid") + if value["binding_changes_applied"] is not False: + raise ContractError("catalog discovery cannot apply binding changes") + if set(normalized["same_id_revision_unknown"]) - set(normalized["changed_model_ids"]): + raise ContractError("catalog revision uncertainty is not changed-model scoped") + if normalized["unqualified_candidate_ids"] != normalized["added_model_ids"]: + raise ContractError("catalog additions must remain unqualified candidates") + if set(normalized["unavailable_profile_ids"]) - set(normalized["affected_profile_ids"]): + raise ContractError("catalog unavailable profiles must be affected") + if (value["profile_scope"] == "unavailable" + and (normalized["affected_profile_ids"] + or normalized["unavailable_profile_ids"])): + raise ContractError("catalog change cannot infer profiles without scope") + return json.loads(canonical_json(normalized)) + + def model_fingerprint(harness: str, version: str | None, model: dict[str, Any]) -> str: stable = {"harness": harness, "version": version, "model": model} return hashlib.sha256(json.dumps(stable, sort_keys=True, separators=(",", ":")).encode()).hexdigest() @@ -109,7 +162,7 @@ def models_by_id(snapshot): hashlib.sha256(canonical_json(previous).encode()).hexdigest() if previous is not None else None ) - return { + return validate_catalog_change({ "schema_version": 1, "harness": current["harness"], "previous_sha256": previous_sha256, @@ -125,7 +178,7 @@ def models_by_id(snapshot): "profile_scope": "provided" if profiles is not None else "unavailable", "unqualified_candidate_ids": added, "binding_changes_applied": False, - } + }) def update_last_good(path: Path, *, harness: str, version: str | None, models: Iterable[dict[str, Any]] | None, complete: bool, error: str | None = None, profiles: Iterable[dict[str, Any]] | None = None) -> dict[str, Any]: diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 7941def..9a08f81 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -206,6 +206,12 @@ def command_profile_binding_change(args: argparse.Namespace) -> tuple[dict, int] )), 0 +def command_profile_binding_fallback(args: argparse.Namespace) -> tuple[dict, int]: + return envelope(data=_service(args).profile_binding_fallback( + _read_json(args.file, "profile binding fallback file"), + )), 0 + + def command_profile_binding_show(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).profile_binding_status(args.alias)), 0 @@ -340,6 +346,7 @@ def parser() -> argparse.ArgumentParser: ("binding-bootstrap", command_profile_binding_bootstrap), ("qualification-add", command_profile_qualification_add), ("binding-change", command_profile_binding_change), + ("binding-fallback", command_profile_binding_fallback), ): operation = profile_sub.add_parser(name) operation.add_argument("--file", required=True) diff --git a/plugin/core/src/devsquad/lifecycle.py b/plugin/core/src/devsquad/lifecycle.py index ccbd9ef..0050b55 100644 --- a/plugin/core/src/devsquad/lifecycle.py +++ b/plugin/core/src/devsquad/lifecycle.py @@ -28,6 +28,11 @@ "expected_binding_version", "qualification_id", "rollback_target", "experiment_id", "evaluation_sha256", "actor", "reason", "evidence_refs", } +CATALOG_FALLBACK_FIELDS = { + "schema_version", "decision_id", "action", "alias", + "expected_binding_version", "catalog_change", "actor", "reason", + "evidence_refs", +} SHA256_LENGTH = 64 @@ -340,6 +345,41 @@ def validate_binding_change(value: dict[str, Any]) -> dict[str, Any]: })) +def validate_catalog_fallback(value: dict[str, Any]) -> dict[str, Any]: + """Validate a CAS rollback driven by complete model-removal evidence.""" + from .catalog import validate_catalog_change + + _exact(value, CATALOG_FALLBACK_FIELDS, "catalog fallback") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("catalog fallback schema_version is invalid") + decision_id = _identifier(value["decision_id"], "decision_id") + alias = _identifier(value["alias"], "catalog fallback alias") + if value["action"] != "rollback": + raise ContractError("catalog fallback action must be rollback") + if (type(value["expected_binding_version"]) is not int + or value["expected_binding_version"] < 1): + raise ContractError("catalog fallback expected version is invalid") + if value["actor"] not in {"human", "guarded_auto"}: + raise ContractError("catalog fallback actor is invalid") + reason = _identifier(value["reason"], "catalog fallback reason") + catalog_change = validate_catalog_change(value["catalog_change"]) + if catalog_change["profile_scope"] != "provided": + raise ContractError("catalog fallback requires profile-scoped evidence") + if not catalog_change["removed_model_ids"]: + raise ContractError("catalog fallback requires a removed model") + return json.loads(canonical_json({ + **value, + "decision_id": decision_id, + "alias": alias, + "reason": reason, + "catalog_change": catalog_change, + "evidence_refs": _strings( + value["evidence_refs"], "catalog fallback evidence_refs", + required=True, + ), + })) + + def validate_recorded_at(value: Any) -> str: return _timestamp(value, "recorded_at") diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 8b73475..388541f 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -922,13 +922,27 @@ def profile_qualification_add( store.close() def profile_binding_change(self, change: dict[str, Any]) -> dict[str, Any]: - from .lifecycle import render_binding_decision_markdown - store = self._store() try: result = store.change_profile_binding(change) finally: store.close() + return self._profile_binding_decision_artifacts(result) + + def profile_binding_fallback(self, change: dict[str, Any]) -> dict[str, Any]: + """Apply a catalog-proven fallback to a prior qualified binding.""" + store = self._store() + try: + result = store.fallback_unavailable_profile_binding(change) + finally: + store.close() + return self._profile_binding_decision_artifacts(result) + + def _profile_binding_decision_artifacts( + self, result: dict[str, Any], + ) -> dict[str, Any]: + from .lifecycle import render_binding_decision_markdown + receipt = result["receipt"] json_content = (canonical_json(receipt) + "\n").encode() markdown_content = render_binding_decision_markdown(receipt).encode() diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 0c68a98..29247f9 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1415,6 +1415,8 @@ def bootstrap_profile_binding( violation = profile_template_violation(profile, normalized_template) if violation is not None: raise ContractError(f"baseline profile violates template: {violation}") + if profile["quality_status"] != "proven": + raise ContractError("baseline profile must already be proven") recorded_at = _authoritative_now(now).isoformat() alias = normalized_template["alias"] self.connection.execute("BEGIN IMMEDIATE") @@ -1855,6 +1857,209 @@ def change_profile_binding( self.connection.execute("ROLLBACK") raise + def fallback_unavailable_profile_binding( + self, change: dict[str, Any], *, now: datetime | None = None, + ) -> dict[str, Any]: + """Roll back an unavailable incumbent to the latest safe predecessor.""" + from .lifecycle import ( + guarded_change_violation, + profile_template_violation, + validate_catalog_fallback, + ) + + request = validate_catalog_fallback(change) + request_json = canonical_json(request) + request_sha256 = hashlib.sha256(request_json.encode()).hexdigest() + recorded_at = _authoritative_now(now).isoformat() + catalog_change = request["catalog_change"] + catalog_change_sha256 = hashlib.sha256( + canonical_json(catalog_change).encode(), + ).hexdigest() + unavailable = set(catalog_change["unavailable_profile_ids"]) + removed_models = set(catalog_change["removed_model_ids"]) + self.connection.execute("BEGIN IMMEDIATE") + try: + existing = self.connection.execute( + "SELECT request_sha256,receipt_json,receipt_sha256,recorded_at " + "FROM binding_decisions WHERE decision_id=?", + (request["decision_id"],), + ).fetchone() + if existing is not None: + if existing["request_sha256"] != request_sha256: + raise ConflictError( + "decision id was already used with a different request", + ) + self.connection.execute("COMMIT") + return { + "receipt": json.loads(existing["receipt_json"]), + "receipt_sha256": existing["receipt_sha256"], + "recorded_at": existing["recorded_at"], + "replayed": True, + } + current = self.connection.execute( + "SELECT b.template_id,b.profile_id,b.qualification_id,b.version," + "p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256 FROM profile_bindings b " + "JOIN concrete_profiles p ON p.profile_id=b.profile_id " + "JOIN profile_templates t ON t.template_id=b.template_id " + "WHERE b.alias=?", + (request["alias"],), + ).fetchone() + if current is None: + raise ContractError("profile binding does not exist") + if current["version"] != request["expected_binding_version"]: + raise ConflictError("profile binding version changed") + current_profile = json.loads(current["profile_json"]) + current_template = json.loads(current["template_json"]) + if (current_profile["id"] not in unavailable + or current_profile["harness"] != catalog_change["harness"] + or current_profile["model_id"] not in removed_models): + raise ContractError( + "catalog evidence does not prove the incumbent unavailable", + ) + if (request["actor"] == "guarded_auto" + and current_template["update_mode"] != "guarded_auto"): + raise ContractError( + "guarded automatic promotion is not enabled", + ) + + target = None + target_profile = None + qualification_payload = None + for candidate in self.connection.execute( + "SELECT v.version,v.template_id,v.profile_id,v.qualification_id," + "p.profile_json,p.profile_sha256,t.payload_json AS template_json," + "t.payload_sha256 AS template_sha256,q.verdict AS qualification_verdict," + "q.payload_json AS qualification_json," + "q.gate_failures_json AS qualification_failures " + "FROM profile_binding_versions v " + "JOIN concrete_profiles p ON p.profile_id=v.profile_id " + "JOIN profile_templates t ON t.template_id=v.template_id " + "LEFT JOIN qualification_runs q " + "ON q.qualification_id=v.qualification_id " + "WHERE v.alias=? AND v.version list[dict[str, Any]]: if not isinstance(alias, str) or not alias: raise ContractError("profile binding alias is invalid") diff --git a/test/core/test_cli.py b/test/core/test_cli.py index cf56c1f..f8bae64 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -212,6 +212,10 @@ def test_profile_lifecycle_commands_dispatch_strict_files(self): "binding-change", "profile_binding_change", "change.json", {"schema_version": 1, "decision_id": "decision-1"}, ), + ( + "binding-fallback", "profile_binding_fallback", "fallback.json", + {"schema_version": 1, "decision_id": "decision-fallback"}, + ), ] for command, method_name, filename, document in operations: with self.subTest(command=command): diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index d42d5b9..bf7d584 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -13,6 +13,7 @@ sys.path.insert(0, str(ROOT / "plugin/core/src")) from devsquad.contracts import ContractError +from devsquad.catalog import update_last_good from devsquad.lifecycle import ( guarded_change_violation, profile_template_violation, @@ -473,6 +474,126 @@ def test_insufficient_evidence_and_disabled_guarded_auto_cannot_promote(self): self.promotion("decision-auto", actor="guarded_auto"), now=NOW, ) + def test_catalog_unavailable_incumbent_uses_only_qualified_predecessor(self): + template = lifecycle_template() + self.store.bootstrap_profile_binding( + template, self.incumbent, version=7, now=NOW, + ) + evaluation = self.seed_experiment() + self.store.record_profile_qualification( + self.qualification(evaluation), now=NOW, + ) + self.store.change_profile_binding( + self.promotion("decision-promote-catalog"), now=NOW, + ) + registry = { + "schema_version": 1, + "profiles": [self.incumbent], + "bindings": { + "review.deep": {"profile_id": "profile-a", "version": 7}, + }, + } + profiles_payload = json.dumps(registry, sort_keys=True) + "\n" + policy_payload = json.dumps(routing_policy(), sort_keys=True) + "\n" + frozen_before = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + self.assertEqual( + load_routing( + review_task(self.repo), frozen_before["profiles_payload"], + policy_payload, + )["roles"]["reviewer"]["selected"]["profile_id"], + "profile-b", + ) + + catalog_path = self.root / "catalog-fallback.json" + update_last_good( + catalog_path, harness="fixture", version="1", complete=True, + models=[{"id": "model-a"}, {"id": "model-b"}], + profiles=[self.incumbent, self.candidate], + ) + catalog_change = update_last_good( + catalog_path, harness="fixture", version="1", complete=True, + models=[{"id": "model-a"}], + profiles=[self.incumbent, self.candidate], + )["catalog_change"] + request = { + "schema_version": 1, + "decision_id": "decision-catalog-fallback", + "action": "rollback", + "alias": "review.deep", + "expected_binding_version": 8, + "catalog_change": catalog_change, + "actor": "guarded_auto", + "reason": "The complete catalog removed the active model.", + "evidence_refs": ["catalog-fallback.json"], + } + service = Service(self.root) + fallback = service.profile_binding_fallback(request) + receipt = fallback["receipt"] + self.assertEqual(receipt["from"]["profile_id"], "profile-b") + self.assertEqual(receipt["to"]["profile_id"], "profile-a") + self.assertEqual(receipt["to"]["binding_version"], 9) + self.assertEqual( + receipt["rollback_evaluation"]["kind"], "catalog_unavailable", + ) + self.assertTrue(receipt["affects_new_runs_only"]) + self.assertTrue(service.profile_binding_fallback(request)["replayed"]) + for artifact in fallback["artifacts"].values(): + content = Path(artifact["path"]).read_bytes() + self.assertEqual(hashlib.sha256(content).hexdigest(), artifact["sha256"]) + + current = self.store.effective_profile_registry( + profiles_payload, policy_payload, + ) + self.assertEqual( + load_routing( + review_task(self.repo), current["profiles_payload"], + policy_payload, + )["roles"]["reviewer"]["selected"]["profile_id"], + "profile-a", + ) + self.assertEqual( + load_routing( + review_task(self.repo), frozen_before["profiles_payload"], + policy_payload, + )["roles"]["reviewer"]["selected"]["profile_id"], + "profile-b", + ) + + self.store.change_profile_binding( + self.promotion("decision-repromote-catalog", expected=9), now=NOW, + ) + all_removed_path = self.root / "catalog-all-removed.json" + update_last_good( + all_removed_path, harness="fixture", version="1", complete=True, + models=[{"id": "model-a"}, {"id": "model-b"}], + profiles=[self.incumbent, self.candidate], + ) + all_removed = update_last_good( + all_removed_path, harness="fixture", version="1", complete=True, + models=[], profiles=[self.incumbent, self.candidate], + )["catalog_change"] + blocked = { + **request, + "decision_id": "decision-catalog-blocked", + "expected_binding_version": 10, + "catalog_change": all_removed, + } + with self.assertRaisesRegex(ContractError, "no available qualified predecessor"): + self.store.fallback_unavailable_profile_binding(blocked, now=NOW) + self.assertEqual( + self.store.profile_binding("review.deep")["profile_id"], "profile-b", + ) + + def test_bootstrap_requires_a_proven_baseline(self): + trial = copy.deepcopy(self.incumbent) + trial["quality_status"] = "trial" + with self.assertRaisesRegex(ContractError, "already be proven"): + self.store.bootstrap_profile_binding( + lifecycle_template(), trial, version=1, now=NOW, + ) + def test_schema_twelve_contains_lifecycle_ledger(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", From 78bf81c7130d69f7a819c0fe3d9dedc76beb5224 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:32:06 -0700 Subject: [PATCH 111/197] docs: checkpoint M6 profile lifecycle --- docs/plans/engineering-team/M6-STATUS.md | 52 +++++++++++++++++++----- docs/plans/engineering-team/RESUME.md | 28 ++++++++----- docs/plans/engineering-team/backlog.json | 9 ++++ 3 files changed, 69 insertions(+), 20 deletions(-) diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md index 46d85a6..da4ae50 100644 --- a/docs/plans/engineering-team/M6-STATUS.md +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -1,9 +1,10 @@ # M6 implementation status M6 is **in progress**. Shared capacity, the append-only outcome ledger, -comparison reports and replay-safe one-variable experiment evaluation are -verified. Draft proposal generation, lifecycle promotion/rollback and the -default-off decision helper remain open. +comparison reports, replay-safe one-variable experiment evaluation, proposal +generation and the complete offline profile lifecycle are verified. The +default-off decision helper and its externally blocked Jev measurement remain +open. | Requirement | Planned evidence | Status | |---|---|---| @@ -15,8 +16,9 @@ default-off decision helper remain open. | Comparison reports | Sample sizes, missingness and separated automatic/pinned/experimental evidence | verified at `45ebc9e` | | Frozen experiment evaluation | One-variable paired evaluation/held-out cases, failure evidence, no-change or promotion-proposal verdict and rollback target; evaluation never changes active policy | verified at `edfb3f3` | | Draft proposals | `learn propose` emits content-addressed JSON/Markdown with hashes, sample sizes, missingness, failures and rollback; no evidence yields no-change | verified at `98c6685` | -| Held-out rerun and rollback | Post-change held-out evidence and exercised rollback through lifecycle bindings | pending | -| Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | pending | +| Held-out rerun and rollback | Post-change held-out evidence and exercised rollback through lifecycle bindings | verified at `4b27e0c` | +| Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | verified through `ca4ee73` | +| Catalog drift and unavailable incumbent | Complete catalog drift scopes revalidation; added models stay unqualified; removed incumbents roll back only to a prior proven/qualified binding or block | verified at `398ae6a` / `ca4ee73` | | Decision helper M6-D1 | Default-off typed contract, fake adapter, cache/accounting and authority/integrity tests | in progress; synthetic Jev probe mechanics only | | Jev M6-D2 | One capped synthetic request with exact model/usage/latency/cost receipt | blocked on `TYPESAFE_API_KEY` | | Laya M6-D3 | Triggered pinned local comparison and measured keep-off/adopt decision | pending; run only if the declared Jev trigger fires | @@ -74,10 +76,40 @@ evidence produces an explicit `no_change`; even qualifying evidence produces only `promotion_proposal`, with `active_policy_changed: false`. The gate is **265 core tests with 2 optional-SDK skips** plus **220/220 Bash assertions**. +## Profile lifecycle checkpoint + +Schema 12 persists strict versioned templates, immutable concrete profiles, +bounded qualification runs, compare-and-swap alias bindings, binding history +and decision receipts. Reviewed and opt-in `guarded_auto` promotion require +saved evaluation and held-out evidence. Guarded changes cannot alter policy, +permissions, billing, account route or expand tools. Binding changes affect +new runs only, while an exact concrete pin and an already frozen run do not +float. + +`4b27e0c` requires a saved post-change `no_change` experiment before ordinary +regression rollback. The experiment must compare the active profile with the +requested predecessor, and its hash, failures, metrics and reasons are retained +in the decision receipt. `398ae6a` records complete catalog drift without +changing a binding: only profiles using changed or removed model IDs are +affected, same-ID drift remains explicitly uncertain, and added models remain +unqualified. + +`ca4ee73` adds `squad profile binding-fallback --file FILE`. Complete, +profile-scoped removal evidence may roll an unavailable incumbent back to the +newest prior profile that is proven and, when applicable, backed by a still- +qualified run under the same lifecycle template. Removed, suspended, trial, +unqualified or authority-widening predecessors are skipped. If no safe +predecessor exists the transaction fails without changing the binding. The +catalog evidence and its hash are retained in the immutable rollback receipt. + +The combined checkpoint gate is **275 core tests with 2 optional-SDK skips** +and ResourceWarning promoted to error, plus **220/220 Bash assertions**. + ## Exact next slice -Implement versioned profile lifecycle bindings, qualification and guarded -promotion/rollback: allowed templates, bounded trials, compare-and-swap -binding versions, new-run-only effects, qualified fallback and immutable -decision receipts. Then exercise a held-out rerun and rollback before wiring -the default-off decision helper. +Implement M6-D1: the default-off typed decision-helper contract, deterministic +fake adapter, call/cache/accounting ledger and off/shadow/advisory authority +tests. It must never widen deterministic routing eligibility or become +permission, spending, acceptance or promotion authority. Keep the one-request +Jev M6-D2 gate blocked until `TYPESAFE_API_KEY` is supplied; run Laya only if +the predeclared fallback trigger fires. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 3022d96..d1d6b0c 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,7 +2,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 27, 2026 +## Current position — September 29, 2026 - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` remains the published runtime @@ -151,9 +151,18 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne `98c6685` adds `squad learn propose --project PATH`, which writes content-addressed local JSON/Markdown drafts from one consistent ledger snapshot, verifies saved evidence hashes and returns explicit no-change when - evidence is absent. The exact gate is 265 core tests with 2 optional-SDK - skips and 220 Bash assertions. Lifecycle bindings/qualification, held-out - rollback and the default-off decision helper remain open. + evidence is absent. +- M6 profile lifecycle is verified through `ca4ee73`. Schema 12 stores + versioned templates, immutable concrete profiles, bounded qualification, + compare-and-swap bindings and immutable decision receipts. Promotions affect + new runs only; exact pins and frozen runs do not float. `4b27e0c` binds + ordinary regression rollback to a saved post-change held-out evaluation. + `398ae6a` scopes complete catalog drift to affected profiles while keeping + additions unqualified, and `ca4ee73` rolls a removed incumbent only to the + newest prior proven/qualified profile under the same template or blocks + without mutation. The exact gate is 275 core tests with 2 optional-SDK skips + and 220 Bash assertions. Only the default-off decision helper and its + external Jev/Laya measurement path remain open in M6. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -180,7 +189,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 265 discovered through M6 learning proposals; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | +| Python core discovery | 275 discovered through the M6 lifecycle/catalog fallback; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -217,11 +226,10 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M6 with lifecycle qualification, compare-and-swap reviewed or - guarded promotion, new-run-only bindings and rollback receipts. Exercise a - held-out rerun/rollback, then add the default-off decision helper. Do not - rebuild the completed M5 offline path, capacity ledger, outcome ledger, - experiment evaluator or proposal generator. +2. Continue M6 at M6-D1 with the default-off typed decision helper, fake + adapter, cache/call accounting and off/shadow/advisory authority tests. Do + not rebuild the completed M5 offline path, capacity/outcome ledger, + experiment evaluator, proposal generator or profile lifecycle. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 5504a25..c882d8f 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -343,6 +343,15 @@ "recorded_at": "2026-09-29T08:50:09Z", "availability": "tracked_tests" }, + { + "kind": "profile_lifecycle_checkpoint", + "revision": "ca4ee73", + "command_or_action": "275 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, focused catalog/lifecycle/CLI tests and a complete-catalog removal fallback", + "outcome": "Schema 12 lifecycle templates, qualification, compare-and-swap promotion, new-run-only binding changes, held-out regression rollback and immutable receipts are verified. Catalog drift revalidates only affected profiles, keeps additions unqualified, and an unavailable incumbent rolls back only to the newest prior proven/qualified binding under the same template or blocks without mutation.", + "artifact": "M6-STATUS.md", + "recorded_at": "2026-09-29T02:31:32-07:00", + "availability": "tracked_tests" + }, { "kind": "early_experiment_checkpoint", "revision": "70e59cb", From e1afbc64e1f88a21caaecce087a32e71b0bd5e19 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:40:09 -0700 Subject: [PATCH 112/197] feat: constrain optional decision guidance --- plugin/core/schemas/policy.schema.json | 6 +- plugin/core/src/devsquad/decision.py | 435 +++++++++++++++++++++++++ plugin/core/src/devsquad/validation.py | 6 +- test/core/fakes/decision_adapter.py | 54 +++ test/core/test_decision_helper.py | 271 +++++++++++++++ 5 files changed, 768 insertions(+), 4 deletions(-) create mode 100644 plugin/core/src/devsquad/decision.py create mode 100644 test/core/fakes/decision_adapter.py create mode 100644 test/core/test_decision_helper.py diff --git a/plugin/core/schemas/policy.schema.json b/plugin/core/schemas/policy.schema.json index c5f8555..76884f2 100644 --- a/plugin/core/schemas/policy.schema.json +++ b/plugin/core/schemas/policy.schema.json @@ -6,10 +6,12 @@ "schema_version":{"const":1},"id":{"type":"string","minLength":1},"version":{"type":"integer","minimum":1}, "roles":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"type":"array","minItems":1,"items":{"$ref":"#/$defs/candidate"}}}, "task_classes":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"enum":["unvalidated","trial","proven","suspended"]}},"require_different_model_for_review":{"type":"boolean"},"prefer_different_harness_for_review":{"type":"boolean"}, - "account_pools":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"$ref":"#/$defs/account_pool"}},"experiment_budget":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"type":"integer","minimum":0}} + "account_pools":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"$ref":"#/$defs/account_pool"}},"experiment_budget":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"type":"integer","minimum":0}},"decision_helper":{"oneOf":[{"type":"object","additionalProperties":false,"required":["schema_version","mode"],"properties":{"schema_version":{"const":1},"mode":{"const":"off"}}},{"$ref":"#/$defs/decision_helper_enabled"}]} }, "$defs":{ "candidate":{"type":"object","additionalProperties":false,"required":["kind","id"],"properties":{"kind":{"enum":["profile","alias"]},"id":{"type":"string","minLength":1}}}, - "account_pool":{"type":"object","additionalProperties":false,"required":["allowed_billing_modes","max_concurrency"],"properties":{"allowed_billing_modes":{"type":"array","minItems":1,"uniqueItems":true,"items":{"enum":["subscription","paid_api"]}},"max_concurrency":{"type":"integer","minimum":1},"unknown_capacity_policy":{"enum":["allow_bounded","block"]}}} + "account_pool":{"type":"object","additionalProperties":false,"required":["allowed_billing_modes","max_concurrency"],"properties":{"allowed_billing_modes":{"type":"array","minItems":1,"uniqueItems":true,"items":{"enum":["subscription","paid_api"]}},"max_concurrency":{"type":"integer","minimum":1},"unknown_capacity_policy":{"enum":["allow_bounded","block"]}}}, + "sha256":{"type":"string","pattern":"^[0-9a-f]{64}$"}, + "decision_helper_enabled":{"type":"object","additionalProperties":false,"required":["schema_version","mode","purpose","adapter","language","min_confidence","gate_evidence_sha256","budget"],"properties":{"schema_version":{"const":1},"mode":{"enum":["shadow","advisory"]},"purpose":{"type":"object","additionalProperties":false,"required":["id","version","question_sha256","rubric_sha256"],"properties":{"id":{"type":"string","minLength":1},"version":{"type":"integer","minimum":1},"question_sha256":{"$ref":"#/$defs/sha256"},"rubric_sha256":{"$ref":"#/$defs/sha256"}}},"adapter":{"type":"object","additionalProperties":false,"required":["id","model","runtime_revision","calibration_version"],"properties":{"id":{"type":"string","minLength":1},"model":{"type":"string","minLength":1},"runtime_revision":{"type":"string","minLength":1},"calibration_version":{"type":["string","null"],"minLength":1}}},"language":{"type":"string","minLength":1},"min_confidence":{"type":"number","minimum":0,"maximum":1},"gate_evidence_sha256":{"type":["string","null"],"pattern":"^[0-9a-f]{64}$"},"budget":{"type":"object","additionalProperties":false,"required":["max_calls","max_input_bytes","wall_seconds","max_cost_usd"],"properties":{"max_calls":{"const":1},"max_input_bytes":{"type":"integer","minimum":1},"wall_seconds":{"type":"integer","minimum":1},"max_cost_usd":{"type":"number","minimum":0}}}}} } } diff --git a/plugin/core/src/devsquad/decision.py b/plugin/core/src/devsquad/decision.py new file mode 100644 index 0000000..f097379 --- /dev/null +++ b/plugin/core/src/devsquad/decision.py @@ -0,0 +1,435 @@ +"""Optional typed decision-helper contracts with no routing authority expansion.""" + +from __future__ import annotations + +import hashlib +import json +import math +from typing import Any + +from .contracts import ContractError +from .store import canonical_json + + +SHA256_LENGTH = 64 +ROLES = {"implementer", "reviewer", "lead", "researcher"} +OFF_FIELDS = {"schema_version", "mode"} +ENABLED_FIELDS = { + "schema_version", "mode", "purpose", "adapter", "language", + "min_confidence", "gate_evidence_sha256", "budget", +} +REQUEST_FIELDS = { + "schema_version", "purpose", "scope_sha256", "evidence", + "candidate_catalog_sha256", "candidates", "pins", "adapter", "language", + "truncation", +} +RESPONSE_FIELDS = { + "schema_version", "request_sha256", "adapter", "language", "truncation", + "recommendations", "usage", "elapsed_ms", +} + + +def _exact(value: Any, fields: set[str], label: str) -> dict[str, Any]: + if not isinstance(value, dict) or set(value) != fields: + raise ContractError(f"{label} fields are invalid") + return value + + +def _identifier(value: Any, field: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ContractError(f"decision helper {field} must be a non-empty string") + return value + + +def _sha256(value: Any, field: str, *, nullable: bool = False) -> str | None: + if value is None and nullable: + return None + if (not isinstance(value, str) or len(value) != SHA256_LENGTH + or any(character not in "0123456789abcdef" for character in value)): + raise ContractError(f"decision helper {field} must be a SHA256 digest") + return value + + +def _finite_probability(value: Any, field: str) -> float: + if (isinstance(value, bool) or not isinstance(value, (int, float)) + or not math.isfinite(value) or not 0.0 <= float(value) <= 1.0): + raise ContractError(f"decision helper {field} must be a finite probability") + return float(value) + + +def _adapter(value: Any) -> dict[str, Any]: + adapter = _exact( + value, {"id", "model", "runtime_revision", "calibration_version"}, + "decision helper adapter", + ) + for field in ("id", "model", "runtime_revision"): + _identifier(adapter[field], f"adapter.{field}") + calibration = adapter["calibration_version"] + if calibration is not None: + _identifier(calibration, "adapter.calibration_version") + return dict(adapter) + + +def validate_decision_policy(value: dict[str, Any] | None) -> dict[str, Any]: + """Validate the optional reviewed policy; omission is exactly mode off.""" + if value is None: + return {"schema_version": 1, "mode": "off"} + if not isinstance(value, dict): + raise ContractError("decision helper policy must be an object") + mode = value.get("mode") + fields = OFF_FIELDS if mode == "off" else ENABLED_FIELDS + _exact(value, fields, "decision helper policy") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("decision helper policy schema_version is invalid") + if mode not in {"off", "shadow", "advisory"}: + raise ContractError("decision helper mode is invalid") + if mode == "off": + return {"schema_version": 1, "mode": "off"} + purpose = _exact( + value["purpose"], + {"id", "version", "question_sha256", "rubric_sha256"}, + "decision helper purpose", + ) + _identifier(purpose["id"], "purpose.id") + if type(purpose["version"]) is not int or purpose["version"] < 1: + raise ContractError("decision helper purpose.version is invalid") + _sha256(purpose["question_sha256"], "purpose.question_sha256") + _sha256(purpose["rubric_sha256"], "purpose.rubric_sha256") + adapter = _adapter(value["adapter"]) + language = _identifier(value["language"], "language") + minimum = _finite_probability(value["min_confidence"], "min_confidence") + gate = _sha256( + value["gate_evidence_sha256"], "gate_evidence_sha256", nullable=True, + ) + if mode == "advisory" and gate is None: + raise ContractError("advisory mode requires reviewed gate evidence") + if mode == "shadow" and gate is not None: + raise ContractError("shadow mode cannot claim advisory gate evidence") + budget = _exact( + value["budget"], + {"max_calls", "max_input_bytes", "wall_seconds", "max_cost_usd"}, + "decision helper budget", + ) + for field in ("max_calls", "max_input_bytes", "wall_seconds"): + if type(budget[field]) is not int or budget[field] < 1: + raise ContractError(f"decision helper budget.{field} is invalid") + if budget["max_calls"] != 1: + raise ContractError("decision helper v1 permits exactly one call") + cost = budget["max_cost_usd"] + if (isinstance(cost, bool) or not isinstance(cost, (int, float)) + or not math.isfinite(cost) or cost < 0): + raise ContractError("decision helper budget.max_cost_usd is invalid") + return json.loads(canonical_json({ + **value, + "purpose": dict(purpose), + "adapter": adapter, + "language": language, + "min_confidence": minimum, + "gate_evidence_sha256": gate, + "budget": dict(budget), + })) + + +def build_decision_request( + task: dict[str, Any], + routing: dict[str, Any], + policy: dict[str, Any], + evidence_payload: bytes | str, + *, + truncated: bool = False, +) -> dict[str, Any] | None: + """Build a cache-complete request containing hashes, never raw task text.""" + config = validate_decision_policy(policy.get("decision_helper")) + if config["mode"] == "off": + return None + if isinstance(evidence_payload, str): + evidence = evidence_payload.encode() + elif isinstance(evidence_payload, bytes): + evidence = evidence_payload + else: + raise ContractError("decision helper evidence payload must be bytes or text") + if len(evidence) > config["budget"]["max_input_bytes"] and not truncated: + raise ContractError("decision helper input exceeds its frozen byte budget") + candidates = {} + pins = {} + for role, routed in routing["roles"].items(): + ordered = [routed["selected"], *routed.get("fallbacks", [])] + identifiers = [candidate["profile_id"] for candidate in ordered] + if len(identifiers) != len(set(identifiers)): + raise ContractError("decision helper candidates must be unique") + candidates[role] = identifiers + if routed["source"] == "override": + pins[role] = routed["selected"]["profile_id"] + request = { + "schema_version": 1, + "purpose": config["purpose"], + "scope_sha256": hashlib.sha256( + canonical_json(task["scope"]).encode(), + ).hexdigest(), + "evidence": { + "sha256": hashlib.sha256(evidence).hexdigest(), + "byte_size": len(evidence), + }, + "candidate_catalog_sha256": routing["profile_registry"]["sha256"], + "candidates": candidates, + "pins": pins, + "adapter": config["adapter"], + "language": config["language"], + "truncation": { + "occurred": bool(truncated), + "limit_bytes": config["budget"]["max_input_bytes"], + }, + } + return validate_decision_request(request) + + +def validate_decision_request(value: dict[str, Any]) -> dict[str, Any]: + _exact(value, REQUEST_FIELDS, "decision helper request") + if type(value["schema_version"]) is not int or value["schema_version"] != 1: + raise ContractError("decision helper request schema_version is invalid") + purpose = _exact( + value["purpose"], + {"id", "version", "question_sha256", "rubric_sha256"}, + "decision helper request purpose", + ) + _identifier(purpose["id"], "request purpose.id") + if type(purpose["version"]) is not int or purpose["version"] < 1: + raise ContractError("decision helper request purpose.version is invalid") + _sha256(purpose["question_sha256"], "request purpose.question_sha256") + _sha256(purpose["rubric_sha256"], "request purpose.rubric_sha256") + _sha256(value["scope_sha256"], "request scope_sha256") + _sha256(value["candidate_catalog_sha256"], "request candidate_catalog_sha256") + evidence = _exact( + value["evidence"], {"sha256", "byte_size"}, "decision helper evidence", + ) + _sha256(evidence["sha256"], "request evidence.sha256") + if type(evidence["byte_size"]) is not int or evidence["byte_size"] < 0: + raise ContractError("decision helper evidence.byte_size is invalid") + candidates = value["candidates"] + if (not isinstance(candidates, dict) or not candidates + or set(candidates) - ROLES): + raise ContractError("decision helper candidates are invalid") + normalized_candidates = {} + for role, identifiers in candidates.items(): + if (not isinstance(identifiers, list) or not identifiers + or len(identifiers) != len(set(identifiers)) + or any(not isinstance(item, str) or not item for item in identifiers)): + raise ContractError("decision helper candidate IDs are invalid") + normalized_candidates[role] = list(identifiers) + pins = value["pins"] + if not isinstance(pins, dict) or set(pins) - set(normalized_candidates): + raise ContractError("decision helper pins are invalid") + for role, profile_id in pins.items(): + if profile_id not in normalized_candidates[role]: + raise ContractError("decision helper pin is outside eligible candidates") + truncation = _exact( + value["truncation"], {"occurred", "limit_bytes"}, + "decision helper request truncation", + ) + if type(truncation["occurred"]) is not bool: + raise ContractError("decision helper truncation flag is invalid") + if type(truncation["limit_bytes"]) is not int or truncation["limit_bytes"] < 1: + raise ContractError("decision helper truncation limit is invalid") + return json.loads(canonical_json({ + **value, + "purpose": dict(purpose), + "evidence": dict(evidence), + "candidates": normalized_candidates, + "pins": dict(pins), + "adapter": _adapter(value["adapter"]), + "language": _identifier(value["language"], "request language"), + "truncation": dict(truncation), + })) + + +def decision_cache_key(request: dict[str, Any]) -> str: + normalized = validate_decision_request(request) + return hashlib.sha256(canonical_json(normalized).encode()).hexdigest() + + +def validate_decision_response( + request: dict[str, Any], + response: dict[str, Any], + config: dict[str, Any], +) -> dict[str, Any]: + """Validate IDs, distributions, usage and provider identity strictly.""" + request = validate_decision_request(request) + config = validate_decision_policy(config) + _exact(response, RESPONSE_FIELDS, "decision helper response") + if type(response["schema_version"]) is not int or response["schema_version"] != 1: + raise ContractError("decision helper response schema_version is invalid") + expected_sha = hashlib.sha256(canonical_json(request).encode()).hexdigest() + if response["request_sha256"] != expected_sha: + raise ContractError("decision helper response does not bind the request") + adapter = _adapter(response["adapter"]) + if adapter != request["adapter"] or adapter != config["adapter"]: + raise ContractError("decision helper observed adapter identity drifted") + language = _identifier(response["language"], "response language") + truncation = _exact( + response["truncation"], {"occurred", "detail"}, + "decision helper response truncation", + ) + if type(truncation["occurred"]) is not bool: + raise ContractError("decision helper response truncation flag is invalid") + if truncation["detail"] is not None: + _identifier(truncation["detail"], "response truncation.detail") + recommendations = response["recommendations"] + if not isinstance(recommendations, dict) or set(recommendations) != set(request["candidates"]): + raise ContractError("decision helper recommendation roles are invalid") + normalized_recommendations = {} + for role, candidate_ids in request["candidates"].items(): + item = _exact( + recommendations[role], + {"ranking", "probabilities", "confidence", "abstain_reason"}, + "decision helper recommendation", + ) + ranking = item["ranking"] + if (not isinstance(ranking, list) or len(ranking) != len(candidate_ids) + or len(ranking) != len(set(ranking)) + or set(ranking) != set(candidate_ids)): + raise ContractError("decision helper ranking changes the eligible set") + probabilities = item["probabilities"] + if not isinstance(probabilities, dict) or set(probabilities) != set(candidate_ids): + raise ContractError("decision helper probability IDs are invalid") + normalized_probabilities = { + profile_id: _finite_probability( + probabilities[profile_id], f"probability.{profile_id}", + ) + for profile_id in candidate_ids + } + if not math.isclose( + sum(normalized_probabilities.values()), 1.0, abs_tol=0.02): + raise ContractError("decision helper probabilities do not sum to one") + confidence = _finite_probability(item["confidence"], "confidence") + abstain = item["abstain_reason"] + if abstain is not None: + abstain = _identifier(abstain, "abstain_reason") + normalized_recommendations[role] = { + "ranking": list(ranking), + "probabilities": normalized_probabilities, + "confidence": confidence, + "abstain_reason": abstain, + } + usage = _exact( + response["usage"], + {"source", "billable_requests", "input_tokens", "output_tokens", "cost_usd"}, + "decision helper usage", + ) + if usage["source"] not in {"native_reported", "fake", "unavailable"}: + raise ContractError("decision helper usage source is invalid") + if (type(usage["billable_requests"]) is not int + or not 0 <= usage["billable_requests"] <= config["budget"]["max_calls"]): + raise ContractError("decision helper billable request count is invalid") + for field in ("input_tokens", "output_tokens"): + if usage[field] is not None and ( + type(usage[field]) is not int or usage[field] < 0): + raise ContractError(f"decision helper usage {field} is invalid") + cost = usage["cost_usd"] + if cost is not None and ( + isinstance(cost, bool) or not isinstance(cost, (int, float)) + or not math.isfinite(cost) or cost < 0 + or cost > config["budget"]["max_cost_usd"]): + raise ContractError("decision helper observed cost exceeds its budget") + if usage["source"] == "unavailable" and any( + usage[field] is not None + for field in ("input_tokens", "output_tokens", "cost_usd")): + raise ContractError("decision helper unavailable usage cannot invent values") + if type(response["elapsed_ms"]) is not int or response["elapsed_ms"] < 0: + raise ContractError("decision helper elapsed_ms is invalid") + return json.loads(canonical_json({ + **response, + "adapter": adapter, + "language": language, + "truncation": dict(truncation), + "recommendations": normalized_recommendations, + "usage": dict(usage), + })) + + +def apply_decision_response( + routing: dict[str, Any], + request: dict[str, Any], + response: dict[str, Any], + config: dict[str, Any], +) -> dict[str, Any]: + """Apply advisory ordering only inside the already-frozen eligible set.""" + config = validate_decision_policy(config) + if config["mode"] == "off": + return json.loads(canonical_json(routing)) + request = validate_decision_request(request) + response = validate_decision_response(request, response, config) + result = json.loads(canonical_json(routing)) + applied_roles = [] + role_status = {} + too_late = response["elapsed_ms"] > config["budget"]["wall_seconds"] * 1000 + truncated = request["truncation"]["occurred"] or response["truncation"]["occurred"] + for role, recommendation in response["recommendations"].items(): + routed = result["roles"][role] + reason = None + if config["mode"] == "shadow": + reason = "shadow_mode" + elif routed["source"] == "override" or role in request["pins"]: + reason = "pinned_route" + elif truncated: + reason = "truncated" + elif too_late: + reason = "late" + elif recommendation["abstain_reason"] is not None: + reason = "abstained" + elif recommendation["confidence"] < config["min_confidence"]: + reason = "below_confidence_gate" + if reason is not None: + role_status[role] = reason + continue + candidates = { + item["profile_id"]: item + for item in [routed["selected"], *routed.get("fallbacks", [])] + } + ordered = [candidates[profile_id] for profile_id in recommendation["ranking"]] + routed["selected"] = ordered[0] + routed["fallbacks"] = ordered[1:] + applied_roles.append(role) + role_status[role] = "applied" + result["decision_helper"] = { + "schema_version": 1, + "mode": config["mode"], + "request_sha256": response["request_sha256"], + "response_sha256": hashlib.sha256( + canonical_json(response).encode(), + ).hexdigest(), + "applied_roles": sorted(applied_roles), + "role_status": role_status, + "usage": response["usage"], + "elapsed_ms": response["elapsed_ms"], + } + return result + + +def decision_fallback( + routing: dict[str, Any], mode: str, status: str, request_sha256: str | None, +) -> dict[str, Any]: + """Record an unusable helper result without changing deterministic routing.""" + if mode == "off": + return json.loads(canonical_json(routing)) + if mode not in {"shadow", "advisory"}: + raise ContractError("decision helper fallback mode is invalid") + status = _identifier(status, "fallback status") + if request_sha256 is not None: + _sha256(request_sha256, "fallback request_sha256") + result = json.loads(canonical_json(routing)) + result["decision_helper"] = { + "schema_version": 1, + "mode": mode, + "request_sha256": request_sha256, + "response_sha256": None, + "applied_roles": [], + "role_status": { + role: status for role in sorted(result["roles"]) + }, + "usage": { + "source": "unavailable", "billable_requests": 0, + "input_tokens": None, "output_tokens": None, "cost_usd": None, + }, + "elapsed_ms": None, + } + return result diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py index 2647879..fc270b7 100644 --- a/plugin/core/src/devsquad/validation.py +++ b/plugin/core/src/devsquad/validation.py @@ -139,8 +139,8 @@ def validate_profile_registry(value: dict[str, Any]) -> None: def validate_policy(value: dict[str, Any]) -> None: - fields = {"schema_version", "id", "version", "roles", "task_classes", "require_different_model_for_review", "prefer_different_harness_for_review", "account_pools", "experiment_budget"} - required = fields - {"prefer_different_harness_for_review"} + fields = {"schema_version", "id", "version", "roles", "task_classes", "require_different_model_for_review", "prefer_different_harness_for_review", "account_pools", "experiment_budget", "decision_helper"} + required = fields - {"prefer_different_harness_for_review", "decision_helper"} _exact(value, fields, required, "policy") if type(value["schema_version"]) is not int or value["schema_version"] != 1 or type(value["version"]) is not int or value["version"] < 1: raise ContractError("invalid policy version") if not isinstance(value["id"], str) or not value["id"]: raise ContractError("policy id must be non-empty") @@ -177,6 +177,8 @@ def validate_policy(value: dict[str, Any]) -> None: raise ContractError("account pool unknown_capacity_policy is invalid") if not all(isinstance(k, str) and k and type(v) is int and v >= 0 for k, v in value["experiment_budget"].items()): raise ContractError("experiment_budget must contain non-negative integers") + from .decision import validate_decision_policy + validate_decision_policy(value.get("decision_helper")) def _validate_json_tree(value: Any, label: str) -> None: diff --git a/test/core/fakes/decision_adapter.py b/test/core/fakes/decision_adapter.py new file mode 100644 index 0000000..4efc218 --- /dev/null +++ b/test/core/fakes/decision_adapter.py @@ -0,0 +1,54 @@ +"""Deterministic typed decision adapter used only by offline core tests.""" + +from __future__ import annotations + +import hashlib +import json + + +def _canonical(value): + return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False) + + +class FakeDecisionAdapter: + def __init__(self, rankings=None, *, confidence=0.9, elapsed_ms=3): + self.rankings = rankings or {} + self.confidence = confidence + self.elapsed_ms = elapsed_ms + self.calls = 0 + + def decide(self, request): + self.calls += 1 + recommendations = {} + for role, candidates in request["candidates"].items(): + ranking = list(self.rankings.get(role, candidates)) + denominator = sum(range(1, len(ranking) + 1)) + descending = list(range(len(ranking), 0, -1)) + scores = { + profile_id: score / denominator + for profile_id, score in zip(ranking, descending) + } + recommendations[role] = { + "ranking": ranking, + "probabilities": scores, + "confidence": self.confidence, + "abstain_reason": None, + } + return { + "schema_version": 1, + "request_sha256": hashlib.sha256( + _canonical(request).encode(), + ).hexdigest(), + "adapter": request["adapter"], + "language": request["language"], + "truncation": {"occurred": False, "detail": None}, + "recommendations": recommendations, + "usage": { + "source": "fake", + "billable_requests": 1, + "input_tokens": 10, + "output_tokens": 2, + "cost_usd": 0.0, + }, + "elapsed_ms": self.elapsed_ms, + } diff --git a/test/core/test_decision_helper.py b/test/core/test_decision_helper.py new file mode 100644 index 0000000..0fc3763 --- /dev/null +++ b/test/core/test_decision_helper.py @@ -0,0 +1,271 @@ +import copy +import hashlib +import json +from pathlib import Path +import sys +import unittest + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +sys.path.insert(0, str(ROOT / "test/core/fakes")) + +from decision_adapter import FakeDecisionAdapter +from devsquad.contracts import ContractError +from devsquad.decision import ( + apply_decision_response, + build_decision_request, + decision_cache_key, + decision_fallback, + validate_decision_policy, + validate_decision_response, +) +from devsquad.router import resolve_routing +from devsquad.validation import validate_policy + + +def profile(profile_id, *, permission="read_only", quality="proven"): + return { + "id": profile_id, + "harness": "fixture", + "model_family": "fixture-family", + "model_id": f"model-{profile_id}", + "effort": {"value": "high", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": permission, + "account_pool_id": "fixture-pool", + "billing_mode": "subscription", + "quality_status": quality, + "evidence_refs": [f"evidence-{profile_id}"], + } + + +def decision_config(mode="shadow"): + return { + "schema_version": 1, + "mode": mode, + "purpose": { + "id": "profile-ranking", + "version": 1, + "question_sha256": "a" * 64, + "rubric_sha256": "b" * 64, + }, + "adapter": { + "id": "fixture", + "model": "fixture-v1", + "runtime_revision": "fixture-runtime-1", + "calibration_version": None, + }, + "language": "en", + "min_confidence": 0.7, + "gate_evidence_sha256": "c" * 64 if mode == "advisory" else None, + "budget": { + "max_calls": 1, + "max_input_bytes": 4096, + "wall_seconds": 2, + "max_cost_usd": 0.01, + }, + } + + +class DecisionHelperTest(unittest.TestCase): + def setUp(self): + self.task = { + "schema_version": 1, + "project": { + "repo_path": "/tmp/decision-fixture", + "base_ref": "base", + "target_ref": "target", + }, + "workflow": "branch-review", + "goal": "Review the bounded change.", + "task_class": "fixture-review", + "acceptance": [{ + "id": "review", "description": "Return a bound review.", + "evidence_kind": "review", + }], + "checks": [], + "scope": {"read_paths": ["src"], "write_paths": []}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "profiles.json", "policy_file": "policy.json", + }, + "budget": { + "wall_seconds": 60, "max_worker_invocations": 2, + "max_revisions": 0, "max_fallbacks_per_step": 1, + }, + "origin": {"surface": "test"}, + } + self.registry = { + "schema_version": 1, + "profiles": [ + profile("review-a"), profile("review-b"), + profile("writer", permission="workspace_write"), + profile("trial", quality="trial"), + ], + "bindings": {}, + } + self.policy = { + "schema_version": 1, + "id": "fixture-policy", + "version": 1, + "roles": {"reviewer": [ + {"kind": "profile", "id": "review-a"}, + {"kind": "profile", "id": "writer"}, + {"kind": "profile", "id": "trial"}, + {"kind": "profile", "id": "review-b"}, + ]}, + "task_classes": {"fixture-review": "proven"}, + "require_different_model_for_review": False, + "prefer_different_harness_for_review": False, + "account_pools": {"fixture-pool": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 2, + "unknown_capacity_policy": "allow_bounded", + }}, + "experiment_budget": {}, + } + self.routing = resolve_routing(self.task, self.registry, self.policy) + + def request(self, mode="shadow", payload="bounded evidence"): + policy = {**self.policy, "decision_helper": decision_config(mode)} + return policy, build_decision_request( + self.task, self.routing, policy, payload, + ) + + def test_default_off_is_exactly_equivalent_and_does_not_build_a_request(self): + validate_policy(self.policy) + self.assertEqual( + validate_decision_policy(None), {"schema_version": 1, "mode": "off"}, + ) + self.assertIsNone(build_decision_request( + self.task, self.routing, self.policy, "ignored", + )) + off = apply_decision_response( + self.routing, + {}, + {}, + {"schema_version": 1, "mode": "off"}, + ) + self.assertEqual(off, self.routing) + self.assertNotIn("decision_helper", off) + + def test_shadow_records_a_valid_suggestion_without_changing_routes(self): + policy, request = self.request("shadow") + adapter = FakeDecisionAdapter({"reviewer": ["review-b", "review-a"]}) + response = adapter.decide(request) + routed = apply_decision_response( + self.routing, request, response, policy["decision_helper"], + ) + self.assertEqual(adapter.calls, 1) + self.assertEqual(routed["roles"], self.routing["roles"]) + self.assertEqual( + routed["decision_helper"]["role_status"], + {"reviewer": "shadow_mode"}, + ) + self.assertEqual(routed["decision_helper"]["applied_roles"], []) + + def test_advisory_can_only_reorder_the_eligible_set(self): + policy, request = self.request("advisory") + self.assertEqual(request["candidates"], { + "reviewer": ["review-a", "review-b"], + }) + response = FakeDecisionAdapter({ + "reviewer": ["review-b", "review-a"], + }).decide(request) + routed = apply_decision_response( + self.routing, request, response, policy["decision_helper"], + ) + reviewer = routed["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "review-b") + self.assertEqual( + [item["profile_id"] for item in reviewer["fallbacks"]], ["review-a"], + ) + self.assertEqual( + {item["profile_id"] for item in reviewer["excluded"]}, + {"writer", "trial"}, + ) + + attacked = copy.deepcopy(response) + attacked["recommendations"]["reviewer"]["ranking"][0] = "writer" + with self.assertRaisesRegex(ContractError, "eligible set"): + validate_decision_response( + request, attacked, policy["decision_helper"], + ) + + def test_pin_unknown_ids_nan_and_identity_drift_fail_closed(self): + self.task["routing"]["overrides"] = { + "reviewer": {"profile_id": "review-a", "fallback": "policy"}, + } + pinned_routing = resolve_routing(self.task, self.registry, self.policy) + policy = {**self.policy, "decision_helper": decision_config("advisory")} + request = build_decision_request( + self.task, pinned_routing, policy, "bounded evidence", + ) + response = FakeDecisionAdapter().decide(request) + routed = apply_decision_response( + pinned_routing, request, response, policy["decision_helper"], + ) + self.assertEqual( + routed["roles"]["reviewer"]["selected"]["profile_id"], "review-a", + ) + self.assertEqual( + routed["decision_helper"]["role_status"]["reviewer"], "pinned_route", + ) + + malformed = copy.deepcopy(response) + malformed["recommendations"]["reviewer"]["probabilities"]["review-a"] = float("nan") + with self.assertRaisesRegex(ContractError, "finite probability"): + validate_decision_response(request, malformed, policy["decision_helper"]) + drifted = copy.deepcopy(response) + drifted["adapter"]["model"] = "fixture-latest" + with self.assertRaisesRegex(ContractError, "identity drifted"): + validate_decision_response(request, drifted, policy["decision_helper"]) + + def test_input_drift_changes_cache_key_and_unusable_outputs_preserve_order(self): + policy, request_a = self.request("advisory", "evidence A") + _, request_b = self.request("advisory", "evidence B") + self.assertNotEqual(decision_cache_key(request_a), decision_cache_key(request_b)) + response = FakeDecisionAdapter( + {"reviewer": ["review-b", "review-a"]}, confidence=0.4, + ).decide(request_a) + routed = apply_decision_response( + self.routing, request_a, response, policy["decision_helper"], + ) + self.assertEqual(routed["roles"], self.routing["roles"]) + self.assertEqual( + routed["decision_helper"]["role_status"]["reviewer"], + "below_confidence_gate", + ) + fallback = decision_fallback( + self.routing, "advisory", "invalid_response", + hashlib.sha256(json.dumps(request_a, sort_keys=True).encode()).hexdigest(), + ) + self.assertEqual(fallback["roles"], self.routing["roles"]) + self.assertEqual(fallback["decision_helper"]["applied_roles"], []) + + def test_advisory_requires_gate_and_budget_or_truncation_cannot_be_hidden(self): + invalid = decision_config("advisory") + invalid["gate_evidence_sha256"] = None + with self.assertRaisesRegex(ContractError, "reviewed gate"): + validate_decision_policy(invalid) + policy = {**self.policy, "decision_helper": decision_config("shadow")} + with self.assertRaisesRegex(ContractError, "byte budget"): + build_decision_request( + self.task, self.routing, policy, "x" * 4097, + ) + request = build_decision_request( + self.task, self.routing, policy, "x" * 4097, truncated=True, + ) + response = FakeDecisionAdapter().decide(request) + routed = apply_decision_response( + self.routing, request, response, policy["decision_helper"], + ) + self.assertEqual(routed["roles"], self.routing["roles"]) + self.assertEqual( + routed["decision_helper"]["role_status"]["reviewer"], "shadow_mode", + ) + + +if __name__ == "__main__": + unittest.main() From 7e83b3aaeff25fa03736501d699b5137ed56877e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:48:22 -0700 Subject: [PATCH 113/197] feat: persist decision helper accounting --- .../migrations/013_decision_observations.sql | 42 ++ plugin/core/src/devsquad/store.py | 429 +++++++++++++++++- test/core/test_capacity.py | 2 +- test/core/test_cli.py | 4 +- test/core/test_decision_store.py | 204 +++++++++ test/core/test_handoff_store.py | 4 +- test/core/test_learning.py | 2 +- test/core/test_lifecycle.py | 2 +- test/core/test_store.py | 8 +- 9 files changed, 685 insertions(+), 12 deletions(-) create mode 100644 plugin/core/src/devsquad/migrations/013_decision_observations.sql create mode 100644 test/core/test_decision_store.py diff --git a/plugin/core/src/devsquad/migrations/013_decision_observations.sql b/plugin/core/src/devsquad/migrations/013_decision_observations.sql new file mode 100644 index 0000000..b494873 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/013_decision_observations.sql @@ -0,0 +1,42 @@ +CREATE TABLE decision_cache ( + cache_key TEXT PRIMARY KEY + CHECK(length(cache_key) = 64 AND cache_key NOT GLOB '*[^0-9a-f]*'), + request_json TEXT NOT NULL, + status TEXT NOT NULL CHECK(status IN ( + 'reserved', 'running', 'succeeded', 'abstained', 'invalid', + 'unavailable', 'indeterminate', 'cancelled' + )), + response_json TEXT, + response_sha256 TEXT + CHECK(response_sha256 IS NULL OR ( + length(response_sha256) = 64 + AND response_sha256 NOT GLOB '*[^0-9a-f]*' + )), + billable_calls INTEGER NOT NULL DEFAULT 0 CHECK(billable_calls IN (0, 1)), + usage_json TEXT, + owner_id TEXT NOT NULL, + error TEXT, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + CHECK( + (status IN ('reserved', 'running') AND response_json IS NULL) + OR status NOT IN ('reserved', 'running') + ) +); + +CREATE TABLE run_decision_observations ( + run_id TEXT NOT NULL REFERENCES runs(id), + purpose_id TEXT NOT NULL, + cache_key TEXT NOT NULL REFERENCES decision_cache(cache_key), + mode TEXT NOT NULL CHECK(mode IN ('shadow', 'advisory')), + applied INTEGER NOT NULL DEFAULT 0 CHECK(applied IN (0, 1)), + effect_json TEXT, + recorded_at TEXT NOT NULL, + PRIMARY KEY(run_id, purpose_id) +); + +CREATE INDEX decision_cache_status +ON decision_cache(status, updated_at, cache_key); + +CREATE INDEX run_decision_observations_cache +ON run_decision_observations(cache_key, run_id); diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 29247f9..18390fe 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 12 +SUPPORTED_SCHEMA_VERSION = 13 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -981,6 +981,433 @@ def active_pool_counts(self) -> dict[str, int]: ) } + @staticmethod + def _decision_observation_payload( + row: sqlite3.Row, *, action: str, replayed: bool, + ) -> dict[str, Any]: + return { + "run_id": row["run_id"], + "purpose_id": row["purpose_id"], + "mode": row["mode"], + "cache_key": row["cache_key"], + "status": row["status"], + "billable_calls": row["billable_calls"], + "request": json.loads(row["request_json"]), + "response": ( + json.loads(row["response_json"]) + if row["response_json"] is not None else None + ), + "response_sha256": row["response_sha256"], + "usage": ( + json.loads(row["usage_json"]) + if row["usage_json"] is not None else None + ), + "error": row["error"], + "applied": bool(row["applied"]), + "effect": ( + json.loads(row["effect_json"]) + if row["effect_json"] is not None else None + ), + "created_at": row["created_at"], + "updated_at": row["updated_at"], + "recorded_at": row["recorded_at"], + "action": action, + "replayed": replayed, + } + + def _decision_observation_row( + self, run_id: str, purpose_id: str, + ) -> sqlite3.Row | None: + return self.connection.execute( + "SELECT o.run_id,o.purpose_id,o.mode,o.applied,o.effect_json," + "o.recorded_at,c.cache_key,c.request_json,c.status,c.response_json," + "c.response_sha256,c.billable_calls,c.usage_json,c.owner_id,c.error," + "c.created_at,c.updated_at FROM run_decision_observations o " + "JOIN decision_cache c ON c.cache_key=o.cache_key " + "WHERE o.run_id=? AND o.purpose_id=?", + (run_id, purpose_id), + ).fetchone() + + def claim_decision_observation( + self, + run_id: str, + mode: str, + request: dict[str, Any], + owner_id: str, + *, + now: datetime | None = None, + ) -> dict[str, Any]: + """Claim at most one decision call or reuse its content-addressed result.""" + from .decision import decision_cache_key, validate_decision_request + + normalized = validate_decision_request(request) + if mode not in {"shadow", "advisory"}: + raise ContractError("decision observation mode is invalid") + if not isinstance(owner_id, str) or not owner_id: + raise ContractError("decision observation owner is invalid") + purpose_id = normalized["purpose"]["id"] + cache_key = decision_cache_key(normalized) + request_json = canonical_json(normalized) + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.connection.execute( + "SELECT state FROM runs WHERE id=?", (run_id,), + ).fetchone() + if run is None: + raise ContractError("decision observation run does not exist") + if run["state"] in TERMINAL_STATES: + raise ConflictError( + "terminal run cannot claim a decision observation", + ) + existing_link = self._decision_observation_row(run_id, purpose_id) + if existing_link is not None: + if (existing_link["cache_key"] != cache_key + or existing_link["mode"] != mode + or existing_link["request_json"] != request_json): + raise ConflictError( + "run decision purpose was already bound differently", + ) + status = existing_link["status"] + action = "cached" + if status == "reserved": + self.connection.execute( + "UPDATE decision_cache SET owner_id=?,updated_at=? " + "WHERE cache_key=? AND status='reserved' AND billable_calls=0", + (owner_id, recorded_at, cache_key), + ) + action = "claimed" + elif status == "running": + if existing_link["owner_id"] == owner_id: + action = "in_flight" + else: + self.connection.execute( + "UPDATE decision_cache SET status='indeterminate'," + "error=?,updated_at=? WHERE cache_key=? AND status='running'", + ( + "prior launched decision call outcome is unknown", + recorded_at, cache_key, + ), + ) + action = "abstain" + elif status not in {"succeeded", "abstained"}: + action = "abstain" + row = self._decision_observation_row(run_id, purpose_id) + self.connection.execute("COMMIT") + return self._decision_observation_payload( + row, action=action, replayed=True, + ) + + cached = self.connection.execute( + "SELECT * FROM decision_cache WHERE cache_key=?", (cache_key,), + ).fetchone() + if cached is None: + self.connection.execute( + "INSERT INTO decision_cache(cache_key,request_json,status," + "billable_calls,owner_id,created_at,updated_at) " + "VALUES(?,?,'reserved',0,?,?,?)", + ( + cache_key, request_json, owner_id, recorded_at, + recorded_at, + ), + ) + action = "claimed" + replayed = False + else: + if cached["request_json"] != request_json: + raise ConflictError( + "decision cache key has conflicting request evidence", + ) + action = ( + "cached" if cached["status"] in {"succeeded", "abstained"} + else "in_flight" if cached["status"] in {"reserved", "running"} + else "abstain" + ) + replayed = True + self.connection.execute( + "INSERT INTO run_decision_observations(run_id,purpose_id,cache_key," + "mode,applied,effect_json,recorded_at) VALUES(?,?,?,?,0,NULL,?)", + (run_id, purpose_id, cache_key, mode, recorded_at), + ) + row = self._decision_observation_row(run_id, purpose_id) + self.connection.execute("COMMIT") + return self._decision_observation_payload( + row, action=action, replayed=replayed, + ) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def launch_decision_call( + self, cache_key: str, owner_id: str, *, now: datetime | None = None, + ) -> dict[str, Any]: + """Fence a possibly billable call before crossing the adapter boundary.""" + if not isinstance(cache_key, str) or len(cache_key) != 64: + raise ContractError("decision cache key is invalid") + if not isinstance(owner_id, str) or not owner_id: + raise ContractError("decision observation owner is invalid") + launched_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT status,billable_calls,owner_id FROM decision_cache " + "WHERE cache_key=?", (cache_key,), + ).fetchone() + if row is None: + raise ContractError("decision cache entry does not exist") + if (row["status"] != "reserved" or row["billable_calls"] != 0 + or row["owner_id"] != owner_id): + raise ConflictError("decision call is not launchable") + self.connection.execute( + "UPDATE decision_cache SET status='running',billable_calls=1," + "updated_at=? WHERE cache_key=? AND status='reserved' " + "AND billable_calls=0 AND owner_id=?", + (launched_at, cache_key, owner_id), + ) + if self.connection.execute("SELECT changes()").fetchone()[0] != 1: + raise ConflictError("decision call launch fence changed") + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, "status": "running", + "billable_calls": 1, "launched_at": launched_at, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def complete_decision_call( + self, + cache_key: str, + owner_id: str, + response: dict[str, Any], + config: dict[str, Any], + *, + now: datetime | None = None, + ) -> dict[str, Any]: + """Persist a valid typed response, or a redacted invalid verdict.""" + from .decision import validate_decision_response + + row = self.connection.execute( + "SELECT request_json FROM decision_cache WHERE cache_key=?", + (cache_key,), + ).fetchone() + if row is None: + raise ContractError("decision cache entry does not exist") + request = json.loads(row["request_json"]) + error = None + try: + normalized = validate_decision_response(request, response, config) + except ContractError as exc: + normalized = None + error = str(exc) + completed_at = _authoritative_now(now).isoformat() + if normalized is None: + status = "invalid" + response_json = None + response_sha256 = None + usage = { + "source": "unavailable", "billable_requests": 1, + "input_tokens": None, "output_tokens": None, "cost_usd": None, + } + else: + response_json = canonical_json(normalized) + response_sha256 = hashlib.sha256(response_json.encode()).hexdigest() + usage = normalized["usage"] + abstained = ( + normalized["truncation"]["occurred"] + or all( + recommendation["abstain_reason"] is not None + for recommendation in normalized["recommendations"].values() + ) + ) + status = "abstained" if abstained else "succeeded" + usage_json = canonical_json(usage) + self.connection.execute("BEGIN IMMEDIATE") + try: + current = self.connection.execute( + "SELECT status,owner_id,billable_calls FROM decision_cache " + "WHERE cache_key=?", (cache_key,), + ).fetchone() + if (current is None or current["status"] != "running" + or current["owner_id"] != owner_id + or current["billable_calls"] != 1): + raise ConflictError("decision call completion is fenced") + self.connection.execute( + "UPDATE decision_cache SET status=?,response_json=?," + "response_sha256=?,usage_json=?,error=?,updated_at=? " + "WHERE cache_key=? AND status='running' AND owner_id=?", + ( + status, response_json, response_sha256, usage_json, error, + completed_at, cache_key, owner_id, + ), + ) + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, + "status": status, + "response": normalized, + "response_sha256": response_sha256, + "usage": usage, + "error": error, + "billable_calls": 1, + "completed_at": completed_at, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def finish_decision_without_call( + self, + cache_key: str, + owner_id: str, + status: str, + reason: str, + *, + now: datetime | None = None, + ) -> dict[str, Any]: + """Record unavailable/cancelled preprocessing without inventing usage.""" + if status not in {"unavailable", "cancelled"}: + raise ContractError("decision no-call status is invalid") + if not isinstance(reason, str) or not reason: + raise ContractError("decision no-call reason is invalid") + recorded_at = _authoritative_now(now).isoformat() + usage = { + "source": "unavailable", "billable_requests": 0, + "input_tokens": None, "output_tokens": None, "cost_usd": None, + } + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT status,owner_id,billable_calls FROM decision_cache " + "WHERE cache_key=?", (cache_key,), + ).fetchone() + if (row is None or row["status"] != "reserved" + or row["owner_id"] != owner_id or row["billable_calls"] != 0): + raise ConflictError("decision no-call completion is fenced") + self.connection.execute( + "UPDATE decision_cache SET status=?,usage_json=?,error=?,updated_at=? " + "WHERE cache_key=? AND status='reserved' AND owner_id=?", + ( + status, canonical_json(usage), reason, recorded_at, + cache_key, owner_id, + ), + ) + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, "status": status, + "billable_calls": 0, "usage": usage, "error": reason, + "recorded_at": recorded_at, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def cancel_decision_observation( + self, cache_key: str, owner_id: str, *, now: datetime | None = None, + ) -> dict[str, Any]: + """Cancel before launch, or preserve uncertainty after a launched call.""" + cancelled_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT status,billable_calls,owner_id FROM decision_cache " + "WHERE cache_key=?", (cache_key,), + ).fetchone() + if row is None: + raise ContractError("decision cache entry does not exist") + if row["owner_id"] != owner_id: + raise ConflictError("decision cancellation owner changed") + if row["status"] not in {"reserved", "running"}: + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, "status": row["status"], + "billable_calls": row["billable_calls"], "replayed": True, + } + status = "cancelled" if row["status"] == "reserved" else "indeterminate" + reason = ( + "decision call cancelled before launch" + if status == "cancelled" + else "decision call cancelled after launch; outcome is unknown" + ) + usage = { + "source": "unavailable", + "billable_requests": row["billable_calls"], + "input_tokens": None, "output_tokens": None, "cost_usd": None, + } + self.connection.execute( + "UPDATE decision_cache SET status=?,usage_json=?,error=?,updated_at=? " + "WHERE cache_key=? AND status IN ('reserved','running')", + ( + status, canonical_json(usage), reason, cancelled_at, + cache_key, + ), + ) + self.connection.execute("COMMIT") + return { + "cache_key": cache_key, "status": status, + "billable_calls": row["billable_calls"], "replayed": False, + } + except Exception: + self.connection.execute("ROLLBACK") + raise + + def record_decision_effect( + self, + run_id: str, + purpose_id: str, + cache_key: str, + effect: dict[str, Any], + *, + now: datetime | None = None, + ) -> dict[str, Any]: + """Bind the frozen routing effect to this run without changing cache data.""" + effect_json = canonical_json(effect) + applied = bool(effect.get("applied_roles")) + recorded_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + row = self.connection.execute( + "SELECT cache_key,applied,effect_json FROM run_decision_observations " + "WHERE run_id=? AND purpose_id=?", + (run_id, purpose_id), + ).fetchone() + if row is None or row["cache_key"] != cache_key: + raise ConflictError("run decision observation binding changed") + if row["effect_json"] is not None: + if (row["effect_json"] != effect_json + or bool(row["applied"]) != applied): + raise ConflictError("run decision effect was recorded differently") + observation = self._decision_observation_row(run_id, purpose_id) + self.connection.execute("COMMIT") + return self._decision_observation_payload( + observation, action="cached", replayed=True, + ) + self.connection.execute( + "UPDATE run_decision_observations SET applied=?,effect_json=?," + "recorded_at=? WHERE run_id=? AND purpose_id=?", + ( + int(applied), effect_json, recorded_at, run_id, purpose_id, + ), + ) + observation = self._decision_observation_row(run_id, purpose_id) + self.connection.execute("COMMIT") + return self._decision_observation_payload( + observation, action="recorded", replayed=False, + ) + except Exception: + self.connection.execute("ROLLBACK") + raise + + def decision_observation( + self, run_id: str, purpose_id: str, + ) -> dict[str, Any] | None: + row = self._decision_observation_row(run_id, purpose_id) + if row is None: + return None + return self._decision_observation_payload( + row, action="read", replayed=True, + ) + def record_outcome( self, run_id: str, diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index 078c0ab..fd6bd9a 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -184,7 +184,7 @@ def test_migration_nine_creates_capacity_ledger(self): version = store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 12) + self.assertEqual(version, 13) tables = { row[0] for row in store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_cli.py b/test/core/test_cli.py index f8bae64..915fd92 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -505,7 +505,7 @@ def build_python(): return candidate return None - def test_installed_wheel_contains_and_applies_migrations_through_twelve(self): + def test_installed_wheel_contains_and_applies_current_migrations(self): build_python = self.build_python() if build_python is None: self.skipTest("offline wheel gate requires setuptools>=68 and wheel; set DEVSQUAD_BUILD_PYTHON") @@ -547,7 +547,7 @@ def test_installed_wheel_contains_and_applies_migrations_through_twelve(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 12 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 13 attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} diff --git a/test/core/test_decision_store.py b/test/core/test_decision_store.py new file mode 100644 index 0000000..e921792 --- /dev/null +++ b/test/core/test_decision_store.py @@ -0,0 +1,204 @@ +import copy +from datetime import datetime, timezone +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +sys.path.insert(0, str(ROOT / "test/core/fakes")) + +from decision_adapter import FakeDecisionAdapter +from devsquad.contracts import ContractError +from devsquad.decision import build_decision_request +from devsquad.store import ConflictError, Store + + +NOW = datetime(2026, 9, 29, 3, 0, tzinfo=timezone.utc) + + +def config(mode="shadow"): + return { + "schema_version": 1, + "mode": mode, + "purpose": { + "id": "profile-ranking", + "version": 1, + "question_sha256": "a" * 64, + "rubric_sha256": "b" * 64, + }, + "adapter": { + "id": "fixture", + "model": "fixture-v1", + "runtime_revision": "fixture-runtime-1", + "calibration_version": None, + }, + "language": "en", + "min_confidence": 0.7, + "gate_evidence_sha256": None, + "budget": { + "max_calls": 1, + "max_input_bytes": 4096, + "wall_seconds": 2, + "max_cost_usd": 0.01, + }, + } + + +class DecisionStoreTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-decision-store-") + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "config", "user.name", "Test"], + check=True, + ) + (self.repo / "README").write_text("fixture\n") + subprocess.run(["git", "-C", str(self.repo), "add", "README"], check=True) + subprocess.run(["git", "-C", str(self.repo), "commit", "-qm", "base"], check=True) + self.store = Store(self.root / "state.sqlite3", self.root / "artifacts") + self.task = {"scope": {"read_paths": ["README"], "write_paths": []}} + self.routing = { + "profile_registry": {"sha256": "d" * 64}, + "roles": {"reviewer": { + "source": "automatic", + "selected": {"profile_id": "profile-a"}, + "fallbacks": [{"profile_id": "profile-b"}], + }}, + } + self.policy = {"decision_helper": config()} + self.request = build_decision_request( + self.task, self.routing, self.policy, "redacted fixture evidence", + ) + + def tearDown(self): + self.store.close() + self.temp.cleanup() + + def run_id(self, key): + return self.store.claim_start( + self.repo, key, {"fixture": key}, f"owner-{key}", + ).run_id + + def test_completed_call_is_cached_across_resume_and_runs(self): + run_one = self.run_id("run-one") + claimed = self.store.claim_decision_observation( + run_one, "shadow", self.request, "decision-owner-1", now=NOW, + ) + self.assertEqual((claimed["action"], claimed["billable_calls"]), ("claimed", 0)) + launched = self.store.launch_decision_call( + claimed["cache_key"], "decision-owner-1", now=NOW, + ) + self.assertEqual(launched["billable_calls"], 1) + adapter = FakeDecisionAdapter({"reviewer": ["profile-b", "profile-a"]}) + completed = self.store.complete_decision_call( + claimed["cache_key"], "decision-owner-1", + adapter.decide(self.request), config(), now=NOW, + ) + self.assertEqual((completed["status"], adapter.calls), ("succeeded", 1)) + + resumed = self.store.claim_decision_observation( + run_one, "shadow", self.request, "decision-owner-2", now=NOW, + ) + self.assertEqual((resumed["action"], resumed["billable_calls"]), ("cached", 1)) + run_two = self.run_id("run-two") + shared = self.store.claim_decision_observation( + run_two, "shadow", self.request, "decision-owner-3", now=NOW, + ) + self.assertEqual((shared["action"], shared["billable_calls"]), ("cached", 1)) + self.assertEqual(shared["response"], completed["response"]) + + effect = { + "mode": "shadow", "applied_roles": [], + "role_status": {"reviewer": "shadow_mode"}, + } + recorded = self.store.record_decision_effect( + run_two, "profile-ranking", shared["cache_key"], effect, now=NOW, + ) + self.assertFalse(recorded["applied"]) + self.assertTrue(self.store.record_decision_effect( + run_two, "profile-ranking", shared["cache_key"], effect, now=NOW, + )["replayed"]) + + def test_launched_unknown_outcome_abstains_without_duplicate_call(self): + run_id = self.run_id("run-crash") + claimed = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner-old", now=NOW, + ) + self.store.launch_decision_call( + claimed["cache_key"], "decision-owner-old", now=NOW, + ) + recovered = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner-new", now=NOW, + ) + self.assertEqual(recovered["status"], "indeterminate") + self.assertEqual(recovered["action"], "abstain") + self.assertEqual(recovered["billable_calls"], 1) + with self.assertRaisesRegex(ConflictError, "not launchable"): + self.store.launch_decision_call( + claimed["cache_key"], "decision-owner-new", now=NOW, + ) + + def test_cancellation_before_launch_records_zero_calls(self): + run_id = self.run_id("run-cancel") + claimed = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner", now=NOW, + ) + cancelled = self.store.cancel_decision_observation( + claimed["cache_key"], "decision-owner", now=NOW, + ) + self.assertEqual((cancelled["status"], cancelled["billable_calls"]), ("cancelled", 0)) + replay = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner-new", now=NOW, + ) + self.assertEqual((replay["action"], replay["billable_calls"]), ("abstain", 0)) + + def test_invalid_response_is_redacted_and_cannot_be_retried(self): + run_id = self.run_id("run-invalid") + claimed = self.store.claim_decision_observation( + run_id, "shadow", self.request, "decision-owner", now=NOW, + ) + self.store.launch_decision_call( + claimed["cache_key"], "decision-owner", now=NOW, + ) + invalid = FakeDecisionAdapter().decide(self.request) + invalid["recommendations"]["reviewer"]["probabilities"]["profile-a"] = float("nan") + completed = self.store.complete_decision_call( + claimed["cache_key"], "decision-owner", invalid, config(), now=NOW, + ) + self.assertEqual(completed["status"], "invalid") + self.assertIsNone(completed["response"]) + self.assertEqual(completed["billable_calls"], 1) + saved = self.store.decision_observation(run_id, "profile-ranking") + self.assertEqual(saved["status"], "invalid") + self.assertIsNone(saved["response"]) + with self.assertRaisesRegex(ConflictError, "not launchable"): + self.store.launch_decision_call( + claimed["cache_key"], "decision-owner", now=NOW, + ) + + def test_schema_thirteen_contains_decision_cache_and_run_links(self): + version = self.store.connection.execute( + "SELECT MAX(version) FROM schema_migrations", + ).fetchone()[0] + self.assertEqual(version, 13) + tables = { + row[0] for row in self.store.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table'", + ) + } + self.assertTrue({"decision_cache", "run_decision_observations"} <= tables) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 2630dd7..fbbea6f 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 12, + 13, ) tables = { row[0] @@ -684,7 +684,7 @@ def test_installed_wheel_applies_schema_four_to_twelve(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 12 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 13 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_learning.py b/test/core/test_learning.py index 601dd04..4515837 100644 --- a/test/core/test_learning.py +++ b/test/core/test_learning.py @@ -429,7 +429,7 @@ def test_current_schema_contains_outcome_ledger(self): store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0], - 12, + 13, ) columns = { row[1] for row in store.connection.execute("PRAGMA table_info(outcomes)") diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index bf7d584..5251a62 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -598,7 +598,7 @@ def test_schema_twelve_contains_lifecycle_ledger(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 12) + self.assertEqual(version, 13) tables = { row[0] for row in self.store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_store.py b/test/core/test_store.py index c04301c..d63d7fe 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -482,8 +482,8 @@ def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): ) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 12) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(13,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(14,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -498,7 +498,7 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 12) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") @@ -515,7 +515,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],12) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],13) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From 87fa9cf541304816e64cda3ef9a0c8550228568d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:56:29 -0700 Subject: [PATCH 114/197] feat: integrate bounded decision guidance --- .../decision-helper-baseline-v1.json | 38 ++++ plugin/core/src/devsquad/decision.py | 2 + plugin/core/src/devsquad/service.py | 176 ++++++++++++++++- plugin/core/src/devsquad/store.py | 67 +++++++ test/core/test_decision_helper.py | 26 +++ test/core/test_decision_store.py | 6 +- test/core/test_service.py | 184 ++++++++++++++++++ 7 files changed, 494 insertions(+), 5 deletions(-) create mode 100644 docs/plans/engineering-team/experiments/decision-helper-baseline-v1.json diff --git a/docs/plans/engineering-team/experiments/decision-helper-baseline-v1.json b/docs/plans/engineering-team/experiments/decision-helper-baseline-v1.json new file mode 100644 index 0000000..07c091b --- /dev/null +++ b/docs/plans/engineering-team/experiments/decision-helper-baseline-v1.json @@ -0,0 +1,38 @@ +{ + "schema_version": 1, + "experiment_id": "devsquad-decision-helper-baseline-v1", + "status": "frozen_offline_baseline", + "purpose": "Validate optional decision-helper mechanics and authority boundaries without claiming routing quality.", + "corpus": { + "path": "jev-pilot-v1.json", + "sha256": "473aed2c431bd1ce8218124a6fe3fd3139fdf869468a72cbc8cf184c428d9822", + "data_class": "synthetic_public_fixture", + "contains_private_content": false, + "case_ids": ["T01", "T02", "T03", "T04", "T05", "T06", "T07", "T08"] + }, + "split": { + "mechanics": ["T01", "T02", "T03", "T04", "T05", "T06"], + "held_out": ["T07", "T08"] + }, + "modes": ["off", "shadow", "advisory"], + "resource_ceiling": { + "max_calls_per_cache_key": 1, + "max_input_bytes": 4096, + "wall_seconds": 2, + "max_cost_usd": 0.01, + "network_calls_in_baseline": 0 + }, + "acceptance": { + "off_mode_route_divergences": 0, + "unauthorized_candidate_selections": 0, + "pin_overrides": 0, + "invalid_output_route_changes": 0, + "duplicate_billable_calls_on_replay": 0, + "raw_private_payloads_persisted": 0 + }, + "adoption": { + "advisory_authorized_by_baseline": false, + "runtime_default": "off", + "reason": "Synthetic mechanics and boundary tests cannot establish production routing quality or benefit." + } +} diff --git a/plugin/core/src/devsquad/decision.py b/plugin/core/src/devsquad/decision.py index f097379..40f723a 100644 --- a/plugin/core/src/devsquad/decision.py +++ b/plugin/core/src/devsquad/decision.py @@ -265,6 +265,8 @@ def validate_decision_response( if adapter != request["adapter"] or adapter != config["adapter"]: raise ContractError("decision helper observed adapter identity drifted") language = _identifier(response["language"], "response language") + if language != request["language"]: + raise ContractError("decision helper response language drifted") truncation = _exact( response["truncation"], {"occurred", "detail"}, "decision helper response truncation", diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 388541f..9a06c13 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -22,6 +22,12 @@ ContractError, ProfileUnsupported, ) +from .decision import ( + apply_decision_response, + build_decision_request, + decision_fallback, + validate_decision_policy, +) from .reports import ( build_early_terminal_reports, build_terminal_reports, @@ -393,6 +399,133 @@ def _decode_claim(value: dict[str, Any]) -> HandoffClaim: action="presented", ) + @staticmethod + def _decision_fixture_response( + request: dict[str, Any], fixture: dict[str, Any], + ) -> dict[str, Any]: + if (not isinstance(fixture, dict) + or set(fixture) != {"rankings", "confidence", "elapsed_ms"} + or not isinstance(fixture["rankings"], dict) + or isinstance(fixture["confidence"], bool) + or not isinstance(fixture["confidence"], (int, float)) + or type(fixture["elapsed_ms"]) is not int + or fixture["elapsed_ms"] < 0): + raise ContractError("internal decision fixture is invalid") + if set(fixture["rankings"]) - set(request["candidates"]): + raise ContractError("internal decision fixture role is invalid") + recommendations = {} + for role, candidates in request["candidates"].items(): + ranking = fixture["rankings"].get(role, candidates) + if not isinstance(ranking, list): + raise ContractError("internal decision fixture ranking is invalid") + denominator = sum(range(1, len(ranking) + 1)) + recommendations[role] = { + "ranking": list(ranking), + "probabilities": { + profile_id: weight / denominator + for profile_id, weight in zip( + ranking, range(len(ranking), 0, -1), + ) + }, + "confidence": fixture["confidence"], + "abstain_reason": None, + } + return { + "schema_version": 1, + "request_sha256": hashlib.sha256( + canonical_json(request).encode(), + ).hexdigest(), + "adapter": request["adapter"], + "language": request["language"], + "truncation": {"occurred": False, "detail": None}, + "recommendations": recommendations, + "usage": { + "source": "fake", "billable_requests": 1, + "input_tokens": 10, "output_tokens": 2, "cost_usd": 0.0, + }, + "elapsed_ms": fixture["elapsed_ms"], + } + + def _apply_optional_decision_helper( + self, + store: Store, + run_id: str, + fencing_token: int, + task: dict[str, Any], + routing: dict[str, Any], + policy: dict[str, Any], + base_oid: str, + target_oid: str, + fixture: dict[str, Any] | None, + ) -> tuple[dict[str, Any], dict[str, Any] | None]: + config = validate_decision_policy(policy.get("decision_helper")) + if config["mode"] == "off": + return routing, None + evidence_payload = canonical_json({ + "base_oid": base_oid, + "target_oid": target_oid, + "goal": task["goal"], + "task_class": task["task_class"], + "acceptance": task["acceptance"], + "checks": task["checks"], + "scope": task["scope"], + }) + try: + request = build_decision_request( + task, routing, policy, evidence_payload, + ) + except ContractError: + return ( + decision_fallback( + routing, config["mode"], "request_invalid", None, + ), + None, + ) + if request is None: # Defensive: enabled modes always build a request. + return routing, None + owner_id = f"preparation:{run_id}:{fencing_token}" + observation = store.claim_decision_observation( + run_id, config["mode"], request, owner_id, + ) + response = observation["response"] + if observation["action"] == "claimed": + if fixture is None: + store.finish_decision_without_call( + observation["cache_key"], owner_id, "unavailable", + "configured decision adapter is unavailable", + ) + observation = store.decision_observation( + run_id, request["purpose"]["id"], + ) + else: + store.launch_decision_call(observation["cache_key"], owner_id) + completed = store.complete_decision_call( + observation["cache_key"], owner_id, + self._decision_fixture_response(request, fixture), config, + ) + response = completed["response"] + observation = store.decision_observation( + run_id, request["purpose"]["id"], + ) + if (observation is not None + and observation["status"] in {"succeeded", "abstained"} + and response is not None): + routed = apply_decision_response(routing, request, response, config) + else: + status = ( + observation["status"] if observation is not None + else "observation_unavailable" + ) + routed = decision_fallback( + routing, config["mode"], status, + hashlib.sha256(canonical_json(request).encode()).hexdigest(), + ) + effect = routed["decision_helper"] + saved = store.record_decision_effect( + run_id, request["purpose"]["id"], observation["cache_key"], effect, + ) + return routed, saved + @staticmethod def _handoff_payload(snapshot: HandoffSnapshot, *, include_packet: bool) -> dict[str, Any]: payload = { @@ -424,6 +557,8 @@ def _resolve_snapshot( internal_review_fixture: dict[str, Any] | None = None, internal_lead_fixture: dict[str, Any] | None = None, internal_implementation_fixture: dict[str, Any] | None = None, + internal_decision_fixture: dict[str, Any] | None = None, + preparation_fencing_token: int | None = None, capacity_store: Store | None = None, ) -> dict[str, Any]: repo = resolved_repo or Path(task["project"]["repo_path"]).resolve(strict=True) @@ -484,6 +619,24 @@ def _resolve_snapshot( }) if project_id is None or run_id is None: raise ContractError("public preflight requires run-owned workspace identity") + if preparation_fencing_token is None: + raise ContractError("public preflight requires a preparation fence") + policy_document = json.loads(config_payloads["policy_file"]) + snapshot["routing"], decision_observation = ( + self._apply_optional_decision_helper( + capacity_store, + run_id, + preparation_fencing_token, + task, + snapshot["routing"], + policy_document, + base_oid, + target_oid, + internal_decision_fixture, + ) + ) + if decision_observation is not None: + snapshot["decision_observation"] = decision_observation if task["workflow"] == "branch-review": snapshot["workspace"] = prepare_review_workspace( repo, @@ -607,6 +760,9 @@ def _continue_preparation( internal_implementation_fixture = submitted.get( "_internal_implementation_fixture" ) + internal_decision_fixture = submitted.get( + "_internal_decision_fixture" + ) store.validate_predecessor(run_id, fencing_token, supersedes_run_id) validated_supersedes_run_id = supersedes_run_id validate_task(task, require_existing_repo=True) @@ -632,6 +788,8 @@ def _continue_preparation( internal_review_fixture=internal_review_fixture, internal_lead_fixture=internal_lead_fixture, internal_implementation_fixture=internal_implementation_fixture, + internal_decision_fixture=internal_decision_fixture, + preparation_fencing_token=fencing_token, capacity_store=store, ) if (task["workflow"] == "issue-delivery" @@ -738,6 +896,7 @@ def start( _internal_review_fixture: dict[str, Any] | None = None, _internal_lead_fixture: dict[str, Any] | None = None, _internal_implementation_fixture: dict[str, Any] | None = None, + _internal_decision_fixture: dict[str, Any] | None = None, ) -> dict[str, Any]: validate_task(task, require_existing_repo=True) if _internal_fake_delay is not None and _internal_review_fixture is not None: @@ -748,6 +907,11 @@ def start( and (_internal_fake_delay is not None or task["workflow"] != "issue-delivery")): raise ContractError("internal implementation fixture requires issue-delivery") + if (_internal_decision_fixture is not None + and _internal_fake_delay is not None): + raise ContractError( + "internal decision fixture requires public preflight", + ) submitted = {"task": task, "supersedes_run_id": supersedes_run_id} if _internal_fake_delay is not None: submitted["_internal_fake_delay"] = _internal_fake_delay @@ -759,6 +923,8 @@ def start( submitted["_internal_implementation_fixture"] = ( _internal_implementation_fixture ) + if _internal_decision_fixture is not None: + submitted["_internal_decision_fixture"] = _internal_decision_fixture store = self._store() try: claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}") @@ -1027,7 +1193,7 @@ def status(self, run_id: str) -> dict[str, Any]: next_action = None else: next_action = None - return { + result = { "run_id": run_id, "state": run["state"], "phase": run["phase"], @@ -1040,6 +1206,10 @@ def status(self, run_id: str) -> dict[str, Any]: "next_action": next_action, "capacity": self._status_capacity(store, run), } + decision_observations = store.decision_observations_for_run(run_id) + if decision_observations: + result["decision_helper"] = decision_observations + return result finally: store.close() def events(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any]: @@ -1066,7 +1236,9 @@ def cancel(self, run_id: str) -> dict[str, Any]: store = self._store() try: run = store.run(run_id) - if run["state"] == "queued" and run["phase"] == "preparing": version = store.cancel_preparing(run_id) + if run["state"] == "queued" and run["phase"] == "preparing": + store.cancel_run_decision_observations(run_id) + version = store.cancel_preparing(run_id) elif run["state"] == "queued" and run["phase"] == "launching": version = store.cancel_launching(run_id) elif run["state"] == "queued" and run["phase"] is None: version = store.cancel_queued(run_id) elif run["state"] == "cancelling" and run["phase"] == "recovery_cleanup": diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 18390fe..d049d06 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1361,6 +1361,8 @@ def record_decision_effect( now: datetime | None = None, ) -> dict[str, Any]: """Bind the frozen routing effect to this run without changing cache data.""" + if not isinstance(effect, dict): + raise ContractError("run decision effect must be an object") effect_json = canonical_json(effect) applied = bool(effect.get("applied_roles")) recorded_at = _authoritative_now(now).isoformat() @@ -1408,6 +1410,71 @@ def decision_observation( row, action="read", replayed=True, ) + def decision_observations_for_run(self, run_id: str) -> list[dict[str, Any]]: + purposes = self.connection.execute( + "SELECT purpose_id FROM run_decision_observations " + "WHERE run_id=? ORDER BY purpose_id", + (run_id,), + ).fetchall() + return [ + self._decision_observation_payload( + self._decision_observation_row(run_id, row["purpose_id"]), + action="read", replayed=True, + ) + for row in purposes + ] + + def cancel_run_decision_observations( + self, run_id: str, *, now: datetime | None = None, + ) -> list[dict[str, Any]]: + """Fence every pending helper call when its owning run is cancelled.""" + cancelled_at = _authoritative_now(now).isoformat() + self.connection.execute("BEGIN IMMEDIATE") + try: + rows = self.connection.execute( + "SELECT DISTINCT c.cache_key,c.status,c.billable_calls " + "FROM run_decision_observations o JOIN decision_cache c " + "ON c.cache_key=o.cache_key WHERE o.run_id=? " + "AND c.status IN ('reserved','running')", + (run_id,), + ).fetchall() + results = [] + for row in rows: + status = ( + "cancelled" if row["status"] == "reserved" + else "indeterminate" + ) + reason = ( + "owning run cancelled before decision launch" + if status == "cancelled" + else "owning run cancelled after decision launch; outcome is unknown" + ) + usage = { + "source": "unavailable", + "billable_requests": row["billable_calls"], + "input_tokens": None, + "output_tokens": None, + "cost_usd": None, + } + self.connection.execute( + "UPDATE decision_cache SET status=?,usage_json=?,error=?," + "updated_at=? WHERE cache_key=? AND status=?", + ( + status, canonical_json(usage), reason, cancelled_at, + row["cache_key"], row["status"], + ), + ) + results.append({ + "cache_key": row["cache_key"], + "status": status, + "billable_calls": row["billable_calls"], + }) + self.connection.execute("COMMIT") + return results + except Exception: + self.connection.execute("ROLLBACK") + raise + def record_outcome( self, run_id: str, diff --git a/test/core/test_decision_helper.py b/test/core/test_decision_helper.py index 0fc3763..d7dd987 100644 --- a/test/core/test_decision_helper.py +++ b/test/core/test_decision_helper.py @@ -266,6 +266,32 @@ def test_advisory_requires_gate_and_budget_or_truncation_cannot_be_hidden(self): routed["decision_helper"]["role_status"]["reviewer"], "shadow_mode", ) + def test_frozen_baseline_uses_only_the_synthetic_labeled_corpus(self): + experiment_dir = ROOT / "docs/plans/engineering-team/experiments" + baseline = json.loads( + (experiment_dir / "decision-helper-baseline-v1.json").read_text(), + ) + self.assertEqual(set(baseline), { + "schema_version", "experiment_id", "status", "purpose", "corpus", + "split", "modes", "resource_ceiling", "acceptance", "adoption", + }) + corpus_path = experiment_dir / baseline["corpus"]["path"] + self.assertEqual( + hashlib.sha256(corpus_path.read_bytes()).hexdigest(), + baseline["corpus"]["sha256"], + ) + corpus = json.loads(corpus_path.read_text()) + case_ids = [case["id"] for case in corpus["cases"]] + self.assertEqual(case_ids, baseline["corpus"]["case_ids"]) + self.assertEqual( + set(baseline["split"]["mechanics"] + baseline["split"]["held_out"]), + set(case_ids), + ) + self.assertFalse(baseline["corpus"]["contains_private_content"]) + self.assertEqual(baseline["resource_ceiling"]["network_calls_in_baseline"], 0) + self.assertFalse(baseline["adoption"]["advisory_authorized_by_baseline"]) + self.assertEqual(baseline["adoption"]["runtime_default"], "off") + if __name__ == "__main__": unittest.main() diff --git a/test/core/test_decision_store.py b/test/core/test_decision_store.py index e921792..294250e 100644 --- a/test/core/test_decision_store.py +++ b/test/core/test_decision_store.py @@ -154,9 +154,9 @@ def test_cancellation_before_launch_records_zero_calls(self): claimed = self.store.claim_decision_observation( run_id, "shadow", self.request, "decision-owner", now=NOW, ) - cancelled = self.store.cancel_decision_observation( - claimed["cache_key"], "decision-owner", now=NOW, - ) + cancelled = self.store.cancel_run_decision_observations( + run_id, now=NOW, + )[0] self.assertEqual((cancelled["status"], cancelled["billable_calls"]), ("cancelled", 0)) replay = self.store.claim_decision_observation( run_id, "shadow", self.request, "decision-owner-new", now=NOW, diff --git a/test/core/test_service.py b/test/core/test_service.py index 85126bf..2398857 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -116,6 +116,67 @@ def wait_state(self, run_id, states, timeout=8): time.sleep(.05) self.fail(f"run did not reach {states}: {self.service.status(run_id)}") + def configure_decision_helper(self, mode): + profiles = json.loads((self.repo / "profiles.json").read_text()) + if not any(item["id"] == "fixture-reviewer-b" for item in profiles["profiles"]): + second = dict(profiles["profiles"][0]) + second.update({ + "id": "fixture-reviewer-b", + "model_id": "fixture-review-model-b", + "evidence_refs": ["tracked-fixture-b"], + }) + profiles["profiles"].append(second) + policy = json.loads((self.repo / "policy.json").read_text()) + policy["roles"]["reviewer"] = [ + {"kind": "alias", "id": "review.deep"}, + {"kind": "profile", "id": "fixture-reviewer-b"}, + ] + policy["decision_helper"] = { + "schema_version": 1, + "mode": mode, + "purpose": { + "id": "profile-ranking", "version": 1, + "question_sha256": "a" * 64, + "rubric_sha256": "b" * 64, + }, + "adapter": { + "id": "fixture", "model": "fixture-v1", + "runtime_revision": "fixture-runtime-1", + "calibration_version": None, + }, + "language": "en", + "min_confidence": 0.7, + "gate_evidence_sha256": "c" * 64 if mode == "advisory" else None, + "budget": { + "max_calls": 1, "max_input_bytes": 4096, + "wall_seconds": 2, "max_cost_usd": 0.01, + }, + } + (self.repo / "profiles.json").write_text( + json.dumps(profiles, sort_keys=True) + "\n", + ) + (self.repo / "policy.json").write_text( + json.dumps(policy, sort_keys=True) + "\n", + ) + subprocess.run( + ["git", "-C", str(self.repo), "add", "profiles.json", "policy.json"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", f"decision {mode}"], + check=True, + ) + + @staticmethod + def decision_fixture(): + return { + "rankings": { + "reviewer": ["fixture-reviewer-b", "fixture-reviewer"], + }, + "confidence": 0.9, + "elapsed_ms": 3, + } + def test_start_is_idempotent_and_result_events_are_durable(self): first=self.service.start(self.task,"same",_internal_fake_delay=.01) second=self.service.start(self.task,"same",_internal_fake_delay=.01) @@ -127,6 +188,129 @@ def test_start_is_idempotent_and_result_events_are_durable(self): self.assertEqual(len(page["events"]),2); self.assertIsNotNone(page["next_cursor"]) with self.assertRaises(ConflictError): self.service.resume(first["run_id"]) + def test_decision_helper_default_off_has_no_runtime_or_call_effect(self): + started = self.service.start( + self.task, "decision-off", + _internal_review_fixture={ + "verdict": "clean", "summary": "off fixture", "findings": [], + }, + ) + self.assertEqual(started["state"], "queued", started) + self.wait_state(started["run_id"], {"awaiting_host"}) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + observations = store.connection.execute( + "SELECT COUNT(*) FROM run_decision_observations WHERE run_id=?", + (started["run_id"],), + ).fetchone()[0] + finally: + store.close() + self.assertNotIn("decision_helper", snapshot["routing"]) + self.assertNotIn("decision_observation", snapshot) + self.assertEqual(observations, 0) + + def test_shadow_decision_is_cached_and_never_changes_selection(self): + self.configure_decision_helper("shadow") + fixture = self.decision_fixture() + run_ids = [] + for key in ("decision-shadow-one", "decision-shadow-two"): + started = self.service.start( + self.task, key, + _internal_review_fixture={ + "verdict": "clean", "summary": key, "findings": [], + }, + _internal_decision_fixture=fixture, + ) + self.assertEqual(started["state"], "queued", started) + self.wait_state(started["run_id"], {"awaiting_host"}) + run_ids.append(started["run_id"]) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + snapshots = [ + json.loads(store.run(run_id)["mutable_snapshot"]) + for run_id in run_ids + ] + observations = [ + store.decision_observation(run_id, "profile-ranking") + for run_id in run_ids + ] + cache_rows = store.connection.execute( + "SELECT COUNT(*) FROM decision_cache", + ).fetchone()[0] + finally: + store.close() + for snapshot in snapshots: + reviewer = snapshot["routing"]["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "fixture-reviewer") + self.assertEqual( + snapshot["routing"]["decision_helper"]["role_status"], + {"reviewer": "shadow_mode"}, + ) + self.assertEqual(observations[0]["cache_key"], observations[1]["cache_key"]) + self.assertEqual([item["billable_calls"] for item in observations], [1, 1]) + self.assertEqual(cache_rows, 1) + self.assertNotIn(self.task["goal"], json.dumps(observations[0]["request"])) + status = self.service.status(run_ids[0]) + self.assertEqual(status["decision_helper"][0]["status"], "succeeded") + self.assertEqual(status["decision_helper"][0]["billable_calls"], 1) + + def test_advisory_reorders_only_eligible_profiles_and_freezes_effect(self): + self.configure_decision_helper("advisory") + started = self.service.start( + self.task, "decision-advisory", + _internal_review_fixture={ + "verdict": "clean", "summary": "advisory fixture", "findings": [], + }, + _internal_decision_fixture=self.decision_fixture(), + ) + self.assertEqual(started["state"], "queued", started) + self.wait_state(started["run_id"], {"awaiting_host"}) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + observation = store.decision_observation( + started["run_id"], "profile-ranking", + ) + finally: + store.close() + reviewer = snapshot["routing"]["roles"]["reviewer"] + self.assertEqual(reviewer["selected"]["profile_id"], "fixture-reviewer-b") + self.assertEqual( + [item["profile_id"] for item in reviewer["fallbacks"]], + ["fixture-reviewer"], + ) + self.assertEqual( + snapshot["routing"]["decision_helper"]["applied_roles"], ["reviewer"], + ) + self.assertTrue(observation["applied"]) + self.assertEqual(observation["billable_calls"], 1) + + def test_missing_decision_adapter_preserves_route_and_records_zero_calls(self): + self.configure_decision_helper("shadow") + started = self.service.start( + self.task, "decision-unavailable", + _internal_review_fixture={ + "verdict": "clean", "summary": "unavailable fixture", "findings": [], + }, + ) + self.assertEqual(started["state"], "queued", started) + self.wait_state(started["run_id"], {"awaiting_host"}) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + observation = store.decision_observation( + started["run_id"], "profile-ranking", + ) + finally: + store.close() + self.assertEqual( + snapshot["routing"]["roles"]["reviewer"]["selected"]["profile_id"], + "fixture-reviewer", + ) + self.assertEqual(observation["status"], "unavailable") + self.assertEqual(observation["billable_calls"], 0) + def test_capacity_observation_drives_preflight_and_status_evidence(self): now = datetime.now(timezone.utc) observed = self.service.capacity_observe({ From d5ec52bdfd3243918723d6f5febf09203ad32b11 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 02:58:00 -0700 Subject: [PATCH 115/197] docs: checkpoint M6 decision helper --- docs/plans/engineering-team/M6-STATUS.md | 46 ++++++++++++++++++------ docs/plans/engineering-team/RESUME.md | 21 ++++++++--- docs/plans/engineering-team/backlog.json | 35 +++++++++++++----- 3 files changed, 79 insertions(+), 23 deletions(-) diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md index da4ae50..efb7d93 100644 --- a/docs/plans/engineering-team/M6-STATUS.md +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -2,9 +2,9 @@ M6 is **in progress**. Shared capacity, the append-only outcome ledger, comparison reports, replay-safe one-variable experiment evaluation, proposal -generation and the complete offline profile lifecycle are verified. The -default-off decision helper and its externally blocked Jev measurement remain -open. +generation, the complete offline profile lifecycle and the default-off typed +decision helper are verified. Only the externally blocked Jev measurement and +its conditional Laya follow-up remain open. | Requirement | Planned evidence | Status | |---|---|---| @@ -19,7 +19,7 @@ open. | Held-out rerun and rollback | Post-change held-out evidence and exercised rollback through lifecycle bindings | verified at `4b27e0c` | | Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | verified through `ca4ee73` | | Catalog drift and unavailable incumbent | Complete catalog drift scopes revalidation; added models stay unqualified; removed incumbents roll back only to a prior proven/qualified binding or block | verified at `398ae6a` / `ca4ee73` | -| Decision helper M6-D1 | Default-off typed contract, fake adapter, cache/accounting and authority/integrity tests | in progress; synthetic Jev probe mechanics only | +| Decision helper M6-D1 | Default-off typed contract, fake adapter, cache/accounting and authority/integrity tests | verified at `87fa9cf` | | Jev M6-D2 | One capped synthetic request with exact model/usage/latency/cost receipt | blocked on `TYPESAFE_API_KEY` | | Laya M6-D3 | Triggered pinned local comparison and measured keep-off/adopt decision | pending; run only if the declared Jev trigger fires | @@ -105,11 +105,37 @@ catalog evidence and its hash are retained in the immutable rollback receipt. The combined checkpoint gate is **275 core tests with 2 optional-SDK skips** and ResourceWarning promoted to error, plus **220/220 Bash assertions**. +## Decision-helper checkpoint + +`e1afbc6` adds a strict optional policy and typed observation/response contract. +Omitting the policy or selecting `off` leaves routing unchanged and creates no +call. `shadow` saves a suggestion without changing execution. `advisory` +requires reviewed gate evidence and can only reorder the exact profiles already +accepted by deterministic permission, billing, capability, identity, quality +and capacity filters. Pins cannot move. Unknown IDs, non-finite scores, +distribution errors, adapter/language drift, truncation, lateness, abstention +and insufficient confidence all preserve deterministic routing. + +Schema 13 at `7e83b3a` stores content-addressed decision requests, per-run +links, response/usage evidence and an explicit billable-call count. The launch +fence is written before an adapter boundary. Resume reuses a completed result; +a call launched before a crash becomes `indeterminate` and cannot be silently +retried. Cancellation before launch records zero calls, invalid output is not +persisted as trusted data, and raw task content is represented only by hashes +and byte counts. + +`87fa9cf` integrates the helper into public preflight and status. The frozen +snapshot records the observation and any advisory effect before worker adapter +selection. Missing optional adapters record `unavailable`, zero calls and the +unchanged route. The deterministic fake adapter proves cache reuse across runs, +shadow equivalence and advisory ordering. The frozen +`decision-helper-baseline-v1.json` links only the synthetic public corpus and +explicitly does not authorize advisory adoption. The gate is **291 core tests +with 2 optional-SDK skips** and **220/220 Bash assertions**. + ## Exact next slice -Implement M6-D1: the default-off typed decision-helper contract, deterministic -fake adapter, call/cache/accounting ledger and off/shadow/advisory authority -tests. It must never widen deterministic routing eligibility or become -permission, spending, acceptance or promotion authority. Keep the one-request -Jev M6-D2 gate blocked until `TYPESAFE_API_KEY` is supplied; run Laya only if -the predeclared fallback trigger fires. +Keep the one-request Jev M6-D2 gate blocked until `TYPESAFE_API_KEY` is +supplied. Do not install or run Laya unless the predeclared Jev access, +cost/usage or measured-quality trigger fires. Continue independent delivery at +M7 packaging, installation, update safety and real-surface usability. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index d1d6b0c..57f86c0 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -163,6 +163,17 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne without mutation. The exact gate is 275 core tests with 2 optional-SDK skips and 220 Bash assertions. Only the default-off decision helper and its external Jev/Laya measurement path remain open in M6. +- M6-D1 is verified at `87fa9cf`. The optional decision helper defaults to off; + shadow records without changing execution, and advisory can only reorder the + deterministic router's already-eligible profiles under reviewed gate + evidence. Schema 13 fences and caches calls, prevents duplicate paid calls on + replay, records launched-unknown outcomes as indeterminate, exposes run + accounting in status and stores only hashes/byte counts for task evidence. + Malformed, unknown-ID, NaN, pin, permission/quality, drift, cancellation and + crash/resume cases fail closed. The frozen synthetic baseline explicitly + keeps runtime adoption off. The gate is 291 core tests with 2 optional-SDK + skips and 220 Bash assertions. M6-D2 remains blocked only on + `TYPESAFE_API_KEY`; Laya remains conditional on its declared trigger. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -189,7 +200,7 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 275 discovered through the M6 lifecycle/catalog fallback; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | +| Python core discovery | 291 discovered through M6-D1 decision-helper integration; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 10 test files, 220 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | @@ -226,10 +237,10 @@ advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M6 at M6-D1 with the default-off typed decision helper, fake - adapter, cache/call accounting and off/shadow/advisory authority tests. Do - not rebuild the completed M5 offline path, capacity/outcome ledger, - experiment evaluator, proposal generator or profile lifecycle. +2. Continue M7 with packaging/install/update migration, fresh standalone use, + update idempotency and active-run survival, compatibility fixtures, + quickstart/troubleshooting and supported local-surface receipts. Do not + rebuild the completed M5 offline path or M6 offline implementation. 3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login or trust blockers are resolved. Their live gates remain open, but M5 may proceed independently from accepted M3. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index c882d8f..d8716e1 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -12,7 +12,7 @@ "execution_brief": "SOL-HANDOFF.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", - "next_milestone": "M5", + "next_milestone": "M7", "milestones": [ { "id": "M1", @@ -281,7 +281,7 @@ { "id": "M6-D1", "title": "Optional typed decision contract and frozen evaluation baseline", - "status": "in_progress", + "status": "complete", "specification": "DECISION-CLASSIFIERS.md", "evidence": [ { @@ -289,12 +289,22 @@ "revision": "70e59cb", "command_or_action": "One-request dry run plus 5 focused probe tests, 231-test complete core discovery with ResourceWarning promoted to error, and 220 Bash assertions", "outcome": "The pinned Jev 1.13 synthetic pilot has 8 cases and 24 typed choices, a $0.01 ceiling, no retry, strict model/probability/usage validation and private-output enforcement; no provider request ran because TypeSafe login/API key is unavailable", - "artifact": "experiments/jev-pilot-v1.json", - "recorded_at": "2026-09-26T17:26:40Z", - "availability": "tracked_fixture_and_tests" - } + "artifact": "experiments/jev-pilot-v1.json", + "recorded_at": "2026-09-26T17:26:40Z", + "availability": "tracked_fixture_and_tests" + }, + { + "kind": "offline_completion_checkpoint", + "revision": "87fa9cf", + "command_or_action": "291 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions and focused contract/store/service decision-helper tests", + "outcome": "Off is a zero-call no-op; shadow is observation-only; advisory is constrained to the deterministic eligible set. Schema 13 provides replay-safe call/cache accounting, launched-unknown fencing, cancellation evidence and public preflight/status integration without persisting raw task text.", + "artifact": "experiments/decision-helper-baseline-v1.json", + "recorded_at": "2026-09-29T02:57:24-07:00", + "availability": "tracked_fixture_and_tests" + } ], - "checkpoint": "One-request synthetic Jev fixture, strict response validator and five offline probe tests are ready; core run/cache integration remains pending" + "checkpoint": "Default-off typed contracts, strict authority bounds, deterministic fake adapter, schema-13 call/cache accounting, crash/cancel/replay fencing, public preflight/status integration and a frozen synthetic baseline are verified at 87fa9cf", + "completed_revision": "87fa9cf" }, { "id": "M6-D2", @@ -352,6 +362,15 @@ "recorded_at": "2026-09-29T02:31:32-07:00", "availability": "tracked_tests" }, + { + "kind": "decision_helper_checkpoint", + "revision": "87fa9cf", + "command_or_action": "291 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 220 Bash assertions, strict off/shadow/advisory authority tests, schema-13 cache/call fencing and public preflight/status integration", + "outcome": "The optional helper defaults to zero-call off mode; shadow cannot alter execution and advisory can only reorder already-eligible unpinned profiles under reviewed gate evidence. Invalid or uncertain output, missing adapters, cancellation and launched-unknown recovery preserve deterministic routing without duplicate paid calls or raw task persistence.", + "artifact": "experiments/decision-helper-baseline-v1.json", + "recorded_at": "2026-09-29T02:57:24-07:00", + "availability": "tracked_fixture_and_tests" + }, { "kind": "early_experiment_checkpoint", "revision": "70e59cb", @@ -362,7 +381,7 @@ "availability": "tracked_fixture_and_tests" } ], - "blocker": null + "blocker": "All independent M6 implementation is complete; the single capped Jev M6-D2 measurement requires TYPESAFE_API_KEY, and M6-D3 runs only if its declared Jev fallback trigger fires" }, { "id": "M7", From 0929867af6927670cf5157fbaa3b89f9674f9bf0 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 03:17:33 -0700 Subject: [PATCH 116/197] feat: add immutable standalone installer --- install.sh | 193 ++++++------- scripts/install-core.sh | 487 +++++++++++++++++++++++++++++++++ test/core/test_install_core.py | 358 ++++++++++++++++++++++++ 3 files changed, 942 insertions(+), 96 deletions(-) create mode 100755 scripts/install-core.sh create mode 100644 test/core/test_install_core.py diff --git a/install.sh b/install.sh index 0da8922..94f4db6 100755 --- a/install.sh +++ b/install.sh @@ -1,117 +1,118 @@ #!/usr/bin/env bash -# DevSquad Installer — registers marketplace, installs plugin, and wires hooks +# DevSquad composite installer: standalone core first, optional Claude plugin. set -euo pipefail -REPO_URL="https://github.com/joshidikshant/devsquad.git" +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd -P)" +REPO_URL="${DEVSQUAD_REPO_URL:-https://github.com/joshidikshant/devsquad.git}" MARKETPLACE="devsquad-marketplace" PLUGIN="devsquad@${MARKETPLACE}" -SETTINGS="$HOME/.claude/settings.json" -PLUGIN_INSTALL_DIR="$HOME/.claude/plugins/marketplaces/${MARKETPLACE}" +CLAUDE_MODE="auto" +STATUS_MODE=0 +JSON_MODE=0 +CORE_ARGS=() + +usage() { + cat <<'EOF' +Usage: ./install.sh [options] + +Installs the standalone DevSquad runtime first. Claude is not required. +When the Claude CLI is available, the legacy plugin is installed or updated +unless --core-only is supplied. + +Composite options: + --core-only Install only the standalone runtime + --with-claude Require and install/update the Claude plugin + +Standalone options are passed to scripts/install-core.sh: + --source-core PATH --install-root PATH --bin-dir PATH + --python PATH --with-mcp --mcp-wheelhouse PATH + --status --json + -h, --help +EOF +} + +fail() { + printf 'devsquad install: %s\n' "$*" >&2 + exit 1 +} + +while [ "$#" -gt 0 ]; do + case "$1" in + --core-only) CLAUDE_MODE="skip"; shift ;; + --with-claude) CLAUDE_MODE="required"; shift ;; + --source-core|--install-root|--bin-dir|--python|--mcp-wheelhouse) + [ "$#" -ge 2 ] || fail "$1 requires a value" + CORE_ARGS+=("$1" "$2"); shift 2 ;; + --with-mcp) + CORE_ARGS+=("$1"); shift ;; + --status) + STATUS_MODE=1; CORE_ARGS+=("$1"); shift ;; + --json) + JSON_MODE=1; CORE_ARGS+=("$1"); shift ;; + -h|--help) usage; exit 0 ;; + *) fail "unknown option: $1" ;; + esac +done + +if [ "$STATUS_MODE" -eq 1 ] && [ "$CLAUDE_MODE" = "required" ]; then + fail "--status cannot be combined with --with-claude" +fi +if [ "$STATUS_MODE" -eq 1 ]; then + CLAUDE_MODE="skip" +fi +if [ "$JSON_MODE" -eq 1 ] && [ "$CLAUDE_MODE" != "skip" ]; then + fail "--json requires --core-only (or --status); use scripts/install-core.sh for standalone JSON" +fi +if [ "$CLAUDE_MODE" = "required" ] && ! command -v claude >/dev/null 2>&1; then + fail "--with-claude requested, but the Claude Code CLI is unavailable" +fi -echo "=== DevSquad Installer ===" -echo +if [ "$JSON_MODE" -eq 1 ]; then + echo "=== DevSquad standalone runtime ===" >&2 +else + echo "=== DevSquad standalone runtime ===" +fi +if [ "${#CORE_ARGS[@]}" -gt 0 ]; then + "$SCRIPT_DIR/scripts/install-core.sh" "${CORE_ARGS[@]}" +else + "$SCRIPT_DIR/scripts/install-core.sh" +fi -# Check claude is available -if ! command -v claude &>/dev/null; then - echo "Error: Claude Code CLI not found. Install it first:" - echo " https://docs.anthropic.com/en/docs/claude-code" - exit 1 +if [ "$CLAUDE_MODE" = "skip" ]; then + if [ "$JSON_MODE" -eq 1 ]; then + echo "Claude plugin: skipped" >&2 + else + echo "Claude plugin: skipped" + fi + exit 0 +fi +if ! command -v claude >/dev/null 2>&1; then + echo "Claude plugin: skipped (Claude Code CLI not found)" + echo "Standalone DevSquad is ready; install the plugin later with ./install.sh --with-claude." + exit 0 fi -# Step 1: Register marketplace -echo "[1/4] Registering marketplace..." +echo +echo "=== DevSquad legacy Claude plugin ===" +echo "[1/3] Marketplace" if claude plugin marketplace list 2>/dev/null | grep -q "$MARKETPLACE"; then - echo " Marketplace already registered, updating..." claude plugin marketplace update "$MARKETPLACE" else claude plugin marketplace add "$REPO_URL" fi -# Step 2: Install plugin -echo "[2/4] Installing plugin..." +echo "[2/3] Plugin" if claude plugin list 2>/dev/null | grep -q "devsquad@"; then - echo " Plugin already installed, updating..." - claude plugin update "$PLUGIN" 2>/dev/null || true + claude plugin update "$PLUGIN" else claude plugin install "$PLUGIN" fi -# Step 3: Enable plugin -echo "[3/4] Enabling plugin..." -claude plugin enable "$PLUGIN" 2>/dev/null || true - -# Step 4: Register hooks into ~/.claude/settings.json (global) -# Hooks point at the MARKETPLACE CLONE (a git checkout that `claude plugin -# marketplace update` refreshes) — never at a versioned cache dir, which -# freezes hooks at install-time and silently drops every later fix. -# Developers hacking on DevSquad itself can point these commands at their -# source checkout instead to run hooks-at-HEAD (see docs/ARCHITECTURE.md). -# Note: per-project hook registration happens during /devsquad:setup (onboarding skill Step 3.5). -echo "[4/4] Registering hooks into global settings.json..." - -if [[ ! -f "$SETTINGS" ]]; then - echo " Creating $SETTINGS..." - echo '{"hooks":{}}' > "$SETTINGS" -fi - -if ! command -v python3 &>/dev/null; then - echo " Warning: python3 not found. Skipping hook registration." - echo " Hooks must be added to $SETTINGS manually." -else - python3 - <&2 + exit 1 +} + +while [ "$#" -gt 0 ]; do + case "$1" in + --source-core) + [ "$#" -ge 2 ] || fail "--source-core requires a path" + SOURCE_CORE="$2"; shift 2 ;; + --install-root) + [ "$#" -ge 2 ] || fail "--install-root requires a path" + INSTALL_ROOT="$2"; shift 2 ;; + --bin-dir) + [ "$#" -ge 2 ] || fail "--bin-dir requires a path" + BIN_DIR="$2"; shift 2 ;; + --python) + [ "$#" -ge 2 ] || fail "--python requires a path" + PYTHON_REQUEST="$2"; shift 2 ;; + --with-mcp) WITH_MCP=1; shift ;; + --mcp-wheelhouse) + [ "$#" -ge 2 ] || fail "--mcp-wheelhouse requires a path" + MCP_WHEELHOUSE="$2"; shift 2 ;; + --status) STATUS_ONLY=1; shift ;; + --json) JSON_OUTPUT=1; shift ;; + -h|--help) usage; exit 0 ;; + *) fail "unknown option: $1" ;; + esac +done + +case "$INSTALL_ROOT" in /*) ;; *) fail "install root must be absolute" ;; esac +case "$BIN_DIR" in /*) ;; *) fail "bin directory must be absolute" ;; esac +[ -d "$SOURCE_CORE/src/devsquad" ] || fail "core source is incomplete: $SOURCE_CORE" +[ -f "$SOURCE_CORE/pyproject.toml" ] || fail "core source has no pyproject.toml: $SOURCE_CORE" + +if [ -x "$PYTHON_REQUEST" ]; then + PYTHON="$PYTHON_REQUEST" +else + PYTHON="$(command -v "$PYTHON_REQUEST" 2>/dev/null || true)" +fi +[ -n "$PYTHON" ] && [ -x "$PYTHON" ] || fail "Python executable not found: $PYTHON_REQUEST" + +PYTHON_INFO="$("$PYTHON" - <<'PY' +import json +import platform +import re +import sys + +if sys.version_info < (3, 11): + raise SystemExit("DevSquad requires Python 3.11+") +cache_tag = sys.implementation.cache_tag or "python" +safe_tag = re.sub(r"[^A-Za-z0-9._-]+", "-", cache_tag) +print(json.dumps({ + "executable": sys.executable, + "version": platform.python_version(), + "version_id": "%d%d%d" % sys.version_info[:3], + "cache_tag": safe_tag, +}, sort_keys=True)) +PY +)" || fail "Python 3.11+ is required" + +core_info() { + "$PYTHON" - "$1" <<'PY' +import hashlib +import json +from pathlib import Path +import re +import sys +import tomllib + +root = Path(sys.argv[1]).resolve(strict=True) +excluded_names = {".DS_Store", ".pytest_cache", "build", "dist", "__pycache__"} +members = [] +for path in root.rglob("*"): + relative = path.relative_to(root) + if any(part in excluded_names or part.endswith(".egg-info") for part in relative.parts): + continue + if path.is_symlink(): + raise SystemExit(f"core source may not contain symlinks: {relative}") + if path.is_file() and path.suffix != ".pyc": + members.append((relative, path)) +digest = hashlib.sha256() +for relative, path in sorted(members, key=lambda item: item[0].as_posix()): + digest.update(relative.as_posix().encode("utf-8") + b"\0") + digest.update(path.read_bytes()) +configuration = tomllib.loads((root / "pyproject.toml").read_text()) +version = configuration["project"]["version"] +init_text = (root / "src/devsquad/__init__.py").read_text() +match = re.search(r'^__version__\s*=\s*["\']([^"\']+)["\']', init_text, re.M) +if not match or match.group(1) != version: + raise SystemExit("pyproject and package versions differ") +if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]*", version): + raise SystemExit("core version is not release-path safe") +print(json.dumps({ + "digest": digest.hexdigest(), + "path": str(root), + "version": version, +}, sort_keys=True)) +PY +} + +SOURCE_INFO="$(core_info "$SOURCE_CORE")" || fail "cannot fingerprint core source" +PLUGIN_CORE="${REPO_ROOT}/plugin/core" +if [ -d "$PLUGIN_CORE/src/devsquad" ]; then + PLUGIN_INFO="$(core_info "$PLUGIN_CORE")" || fail "cannot fingerprint plugin core" +else + PLUGIN_INFO='null' +fi + +emit_report() { + "$PYTHON" - "$1" "$SOURCE_INFO" "$PLUGIN_INFO" "$INSTALL_ROOT" "$BIN_DIR" "$PYTHON_INFO" "$2" "$3" <<'PY' +import hashlib +import json +from pathlib import Path +import sys + +mode, source_raw, plugin_raw, install_raw, bin_raw, python_raw, changed_raw, error = sys.argv[1:] +source = json.loads(source_raw) +plugin = json.loads(plugin_raw) +python = json.loads(python_raw) +install_root = Path(install_raw) +current = install_root / "current" +launcher = Path(bin_raw) / "squad" +installed = None +installed_payload_digest = None +current_target = None +manifest_error = None + +def payload_digest(root): + excluded_names = {".DS_Store", ".pytest_cache", "build", "dist", "__pycache__"} + members = [] + for path in root.rglob("*"): + relative = path.relative_to(root) + if any(part in excluded_names or part.endswith(".egg-info") for part in relative.parts): + continue + if path.is_symlink(): + raise OSError(f"installed core contains a symlink: {relative}") + if path.is_file() and path.suffix != ".pyc": + members.append((relative, path)) + digest = hashlib.sha256() + for relative, path in sorted(members, key=lambda item: item[0].as_posix()): + digest.update(relative.as_posix().encode("utf-8") + b"\0") + digest.update(path.read_bytes()) + return digest.hexdigest() + +try: + if current.is_symlink(): + current_target = str(current.resolve(strict=True)) + installed = json.loads((current / "release.json").read_text()) + installed_payload_digest = payload_digest(current / "core") + if installed.get("source_digest") != installed_payload_digest: + manifest_error = "installed release payload differs from its manifest" + elif current.exists(): + manifest_error = "current selector is not a symlink" +except (OSError, UnicodeError, json.JSONDecodeError) as exc: + manifest_error = f"cannot read installed release: {type(exc).__name__}" +source_digest = source["digest"] +plugin_digest = plugin["digest"] if plugin else None +installed_digest = installed_payload_digest +report = { + "schema_version": 1, + "mode": mode, + "changed": changed_raw == "1", + "source": source, + "plugin": plugin, + "installed": installed, + "installed_payload_digest": installed_payload_digest, + "installed_manifest_matches": bool( + installed is not None + and installed.get("source_digest") == installed_payload_digest + ), + "current_target": current_target, + "launcher": str(launcher), + "launcher_ready": launcher.is_file() and launcher.stat().st_mode & 0o111 != 0, + "python": python, + "drift": { + "source_plugin": plugin_digest is not None and source_digest != plugin_digest, + "source_installed": installed_digest is None or source_digest != installed_digest, + "plugin_installed": plugin_digest is None or installed_digest is None or plugin_digest != installed_digest, + }, + "error": error or manifest_error, +} +print(json.dumps(report, sort_keys=True, separators=(",", ":"))) +PY +} + +if [ "$STATUS_ONLY" -eq 1 ]; then + REPORT="$(emit_report status 0 '')" + if [ "$JSON_OUTPUT" -eq 1 ]; then + printf '%s\n' "$REPORT" + else + "$PYTHON" - "$REPORT" <<'PY' +import json, sys +r = json.loads(sys.argv[1]) +state = "not installed" if r["installed"] is None else r["installed"]["release_id"] +print(f"DevSquad standalone: {state}") +print(f" launcher: {r['launcher']} ({'ready' if r['launcher_ready'] else 'missing'})") +print(" drift: source/plugin={source_plugin} source/installed={source_installed} plugin/installed={plugin_installed}".format(**r["drift"])) +if r["error"]: + print(f" error: {r['error']}") +PY + fi + exit 0 +fi + +if [ "$WITH_MCP" -eq 1 ]; then + [ -n "$MCP_WHEELHOUSE" ] || fail "--with-mcp requires --mcp-wheelhouse" + [ -d "$MCP_WHEELHOUSE" ] || fail "MCP wheelhouse is not a directory: $MCP_WHEELHOUSE" +fi + +mkdir -p "$INSTALL_ROOT/releases" "$BIN_DIR" +INSTALL_ROOT="$(cd "$INSTALL_ROOT" && pwd -P)" +BIN_DIR="$(cd "$BIN_DIR" && pwd -P)" +SOURCE_CORE="$(cd "$SOURCE_CORE" && pwd -P)" + +LOCK_DIR="$INSTALL_ROOT/.install.lock" +if ! mkdir "$LOCK_DIR" 2>/dev/null; then + fail "another install is active or left $LOCK_DIR behind" +fi +TEMP_DIR="" +cleanup() { + if [ -n "$TEMP_DIR" ] && [ -d "$TEMP_DIR" ]; then + rm -rf "$TEMP_DIR" + fi + rmdir "$LOCK_DIR" 2>/dev/null || true +} +trap cleanup EXIT HUP INT TERM + +SOURCE_DIGEST="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["digest"])' "$SOURCE_INFO")" +SOURCE_VERSION="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["version"])' "$SOURCE_INFO")" +PYTHON_VERSION_ID="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["version_id"])' "$PYTHON_INFO")" +PYTHON_CACHE_TAG="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["cache_tag"])' "$PYTHON_INFO")" +FLAVOR="core" +MCP_ENV_DIGEST="" +if [ "$WITH_MCP" -eq 1 ]; then + MCP_ENV_DIGEST="$($PYTHON - "$SOURCE_CORE/requirements-mcp.lock" "$MCP_WHEELHOUSE" <<'PY' +import hashlib +from pathlib import Path +import sys + +digest = hashlib.sha256() +for root in (Path(sys.argv[1]), Path(sys.argv[2])): + paths = [root] if root.is_file() else sorted(p for p in root.rglob("*") if p.is_file()) + for path in paths: + digest.update(path.name.encode("utf-8") + b"\0" + path.read_bytes()) +print(digest.hexdigest()) +PY +)" + FLAVOR="mcp-${MCP_ENV_DIGEST%${MCP_ENV_DIGEST#????????????}}" +fi +SHORT_DIGEST="${SOURCE_DIGEST%${SOURCE_DIGEST#????????????}}" +RELEASE_ID="${SOURCE_VERSION}-py${PYTHON_VERSION_ID}-${SHORT_DIGEST}-${FLAVOR}" +RELEASE_DIR="$INSTALL_ROOT/releases/$RELEASE_ID" +RELEASE_CREATED=0 + +if [ -e "$RELEASE_DIR" ]; then + [ -d "$RELEASE_DIR" ] && [ -f "$RELEASE_DIR/release.json" ] || fail "release path is incomplete: $RELEASE_DIR" + RELEASE_CORE_INFO="$(core_info "$RELEASE_DIR/core")" || fail "existing release payload is unreadable" + RELEASE_CORE_DIGEST="$($PYTHON -c 'import json,sys; print(json.loads(sys.argv[1])["digest"])' "$RELEASE_CORE_INFO")" + [ "$RELEASE_CORE_DIGEST" = "$SOURCE_DIGEST" ] || fail "existing release payload digest differs: $RELEASE_DIR" + "$PYTHON" - "$RELEASE_DIR/release.json" "$RELEASE_ID" "$SOURCE_DIGEST" "$PYTHON_CACHE_TAG" "$WITH_MCP" "$MCP_ENV_DIGEST" <<'PY' +import json +from pathlib import Path +import sys + +path, release_id, digest, cache_tag, with_mcp, mcp_digest = sys.argv[1:] +value = json.loads(Path(path).read_text()) +expected = { + "release_id": release_id, + "source_digest": digest, + "python_cache_tag": cache_tag, + "mcp": with_mcp == "1", + "mcp_environment_digest": mcp_digest or None, +} +for key, item in expected.items(): + if value.get(key) != item: + raise SystemExit(f"existing release manifest mismatch: {key}") +venv_python = Path(path).parent / "venv/bin/python" +if not venv_python.is_file(): + raise SystemExit("existing release has no runtime Python") +PY +else + TEMP_DIR="$(mktemp -d "$INSTALL_ROOT/.install.XXXXXX")" + BUILD_RELEASE="$TEMP_DIR/release" + mkdir -p "$BUILD_RELEASE" + "$PYTHON" -m venv "$BUILD_RELEASE/venv" + VENV_PYTHON="$BUILD_RELEASE/venv/bin/python" + [ -x "$VENV_PYTHON" ] || fail "virtual environment did not provide bin/python" + PURELIB="$($VENV_PYTHON -c 'import sysconfig; print(sysconfig.get_path("purelib"))')" + "$VENV_PYTHON" - "$SOURCE_CORE" "$BUILD_RELEASE/core" "$PURELIB" "$SOURCE_VERSION" <<'PY' +from pathlib import Path +import os +import shutil +import sys + +source = Path(sys.argv[1]) +installed_core = Path(sys.argv[2]) +purelib = Path(sys.argv[3]) +version = sys.argv[4] +ignore = shutil.ignore_patterns( + "__pycache__", "*.pyc", ".DS_Store", ".pytest_cache", "build", "dist", "*.egg-info", +) +shutil.copytree(source, installed_core, ignore=ignore) +relative_source = os.path.relpath(installed_core / "src", purelib) +(purelib / "devsquad-core.pth").write_text(relative_source + "\n") +dist = purelib / f"devsquad_core-{version}.dist-info" +dist.mkdir() +(dist / "METADATA").write_text( + "Metadata-Version: 2.1\nName: devsquad-core\nVersion: " + version + "\n" +) +(dist / "WHEEL").write_text( + "Wheel-Version: 1.0\nGenerator: devsquad-install-core\nRoot-Is-Purelib: true\nTag: py3-none-any\n" +) +(dist / "entry_points.txt").write_text("[console_scripts]\nsquad = devsquad.cli:main\n") +(dist / "RECORD").write_text("") +PY + if [ "$WITH_MCP" -eq 1 ]; then + "$VENV_PYTHON" -m pip install --disable-pip-version-check --no-index \ + --only-binary=:all: --find-links "$MCP_WHEELHOUSE" \ + -r "$SOURCE_CORE/requirements-mcp.lock" + "$VENV_PYTHON" - <<'PY' +from importlib import metadata +import mcp +assert metadata.version("mcp") == "2.2.0" +PY + fi + "$VENV_PYTHON" -P "$BUILD_RELEASE/core/bin/squad" --version >/dev/null + "$VENV_PYTHON" -P - <<'PY' +from devsquad.integrations import load_integrations +assert {item.id for item in load_integrations()} == {"codex", "claude-code", "antigravity", "grok"} +PY + "$PYTHON" - "$BUILD_RELEASE/release.json" "$RELEASE_ID" "$SOURCE_VERSION" "$SOURCE_DIGEST" "$PYTHON_INFO" "$WITH_MCP" "$MCP_ENV_DIGEST" <<'PY' +import json +from pathlib import Path +import sys + +path, release_id, version, digest, python_raw, with_mcp, mcp_digest = sys.argv[1:] +python = json.loads(python_raw) +value = { + "schema_version": 1, + "release_id": release_id, + "version": version, + "source_digest": digest, + "python_version": python["version"], + "python_cache_tag": python["cache_tag"], + "mcp": with_mcp == "1", + "mcp_environment_digest": mcp_digest or None, +} +Path(path).write_text(json.dumps(value, sort_keys=True, separators=(",", ":")) + "\n") +PY + mv "$BUILD_RELEASE" "$RELEASE_DIR" + RELEASE_CREATED=1 +fi + +CURRENT_CHANGED=0 +CURRENT_TARGET="releases/$RELEASE_ID" +if [ -L "$INSTALL_ROOT/current" ] && [ "$(readlink "$INSTALL_ROOT/current")" = "$CURRENT_TARGET" ]; then + : +elif [ -e "$INSTALL_ROOT/current" ] || [ -L "$INSTALL_ROOT/current" ]; then + [ -L "$INSTALL_ROOT/current" ] || fail "current selector is not a symlink: $INSTALL_ROOT/current" + ln -s "$CURRENT_TARGET" "$INSTALL_ROOT/.current.$$" + "$PYTHON" - "$INSTALL_ROOT/.current.$$" "$INSTALL_ROOT/current" <<'PY' +import os, sys +os.replace(sys.argv[1], sys.argv[2]) +PY + CURRENT_CHANGED=1 +else + ln -s "$CURRENT_TARGET" "$INSTALL_ROOT/.current.$$" + "$PYTHON" - "$INSTALL_ROOT/.current.$$" "$INSTALL_ROOT/current" <<'PY' +import os, sys +os.replace(sys.argv[1], sys.argv[2]) +PY + CURRENT_CHANGED=1 +fi + +LAUNCHER="$BIN_DIR/squad" +if [ -e "$LAUNCHER" ] && ! grep -q '^# managed-by: devsquad-install-core-v1$' "$LAUNCHER" 2>/dev/null; then + fail "refusing to replace unmanaged launcher: $LAUNCHER" +fi +LAUNCHER_TEMP="$BIN_DIR/.squad.$$" +"$PYTHON" - "$LAUNCHER_TEMP" "$INSTALL_ROOT" <<'PY' +from pathlib import Path +import shlex +import sys + +path = Path(sys.argv[1]) +python = Path(sys.argv[2]) / "current/venv/bin/python" +entry = Path(sys.argv[2]) / "current/core/bin/squad" +path.write_text( + "#!/bin/sh\n" + "# managed-by: devsquad-install-core-v1\n" + "exec " + shlex.quote(str(python)) + " -P " + shlex.quote(str(entry)) + " \"$@\"\n" +) +path.chmod(0o755) +PY +LAUNCHER_CHANGED=0 +if [ -f "$LAUNCHER" ] && cmp -s "$LAUNCHER_TEMP" "$LAUNCHER"; then + rm -f "$LAUNCHER_TEMP" +else + mv -f "$LAUNCHER_TEMP" "$LAUNCHER" + LAUNCHER_CHANGED=1 +fi + +CHANGED=0 +if [ "$RELEASE_CREATED" -eq 1 ] || [ "$CURRENT_CHANGED" -eq 1 ] || [ "$LAUNCHER_CHANGED" -eq 1 ]; then + CHANGED=1 +fi +REPORT="$(emit_report install "$CHANGED" '')" +STATE_TEMP="$INSTALL_ROOT/.install-state.$$" +"$PYTHON" - "$STATE_TEMP" "$REPORT" <<'PY' +import json +from pathlib import Path +import sys + +path = Path(sys.argv[1]) +report = json.loads(sys.argv[2]) +state = { + "schema_version": report["schema_version"], + "source": report["source"], + "plugin": report["plugin"], + "installed": report["installed"], + "installed_payload_digest": report["installed_payload_digest"], + "installed_manifest_matches": report["installed_manifest_matches"], + "current_target": report["current_target"], + "launcher": report["launcher"], + "python": report["python"], + "drift": report["drift"], +} +path.write_text(json.dumps(state, sort_keys=True, separators=(",", ":")) + "\n") +PY +if [ -f "$INSTALL_ROOT/install-state.json" ] && cmp -s "$STATE_TEMP" "$INSTALL_ROOT/install-state.json"; then + rm -f "$STATE_TEMP" +else + mv -f "$STATE_TEMP" "$INSTALL_ROOT/install-state.json" +fi + +if [ "$JSON_OUTPUT" -eq 1 ]; then + printf '%s\n' "$REPORT" +else + "$PYTHON" - "$REPORT" <<'PY' +import json, sys +r = json.loads(sys.argv[1]) +print(f"DevSquad {r['installed']['version']} installed") +print(f" release: {r['current_target']}") +print(f" launcher: {r['launcher']}") +print(f" changed: {str(r['changed']).lower()}") +print(" drift: source/plugin={source_plugin} source/installed={source_installed} plugin/installed={plugin_installed}".format(**r["drift"])) +PY +fi diff --git a/test/core/test_install_core.py b/test/core/test_install_core.py new file mode 100644 index 0000000..18f2391 --- /dev/null +++ b/test/core/test_install_core.py @@ -0,0 +1,358 @@ +import json +import os +from pathlib import Path +import shutil +import sqlite3 +import subprocess +import sys +import tempfile +import time +import unittest + +from devsquad_test_fixtures import branch_review_routing_documents + + +ROOT = Path(__file__).resolve().parents[2] +CORE = ROOT / "plugin/core" +INSTALLER = ROOT / "scripts/install-core.sh" +COMPOSITE_INSTALLER = ROOT / "install.sh" + + +class StandaloneInstallerTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-install-") + self.root = Path(self.temp.name) + self.install_root = self.root / "installed" + self.bin_dir = self.root / "bin" + self.environment = os.environ.copy() + self.environment.update({ + "DEVSQUAD_INSTALL_ROOT": str(self.install_root), + "DEVSQUAD_BIN_DIR": str(self.bin_dir), + "DEVSQUAD_PYTHON": sys.executable, + }) + self.environment.pop("PYTHONPATH", None) + + def tearDown(self): + self.temp.cleanup() + + def install(self, source=CORE, *arguments): + result = subprocess.run( + [ + "/bin/bash", str(INSTALLER), + "--source-core", str(source), + "--json", *arguments, + ], + check=True, + text=True, + capture_output=True, + env=self.environment, + cwd=ROOT, + ) + self.assertEqual(result.stderr, "") + lines = result.stdout.splitlines() + self.assertEqual(len(lines), 1, result.stdout) + return json.loads(lines[0]) + + def changed_source(self): + destination = self.root / "core-v2" + shutil.copytree( + CORE, + destination, + ignore=shutil.ignore_patterns( + "__pycache__", "*.pyc", ".pytest_cache", "*.egg-info", + ), + ) + pyproject = destination / "pyproject.toml" + pyproject.write_text( + pyproject.read_text().replace('version = "0.1.0"', 'version = "0.1.1"') + ) + package = destination / "src/devsquad/__init__.py" + package.write_text( + package.read_text().replace('__version__ = "0.1.0"', '__version__ = "0.1.1"') + ) + return destination + + def test_fresh_install_requires_no_claude_and_reinstall_is_idempotent(self): + self.environment["PATH"] = "/usr/bin:/bin" + + first = self.install() + self.assertTrue(first["changed"]) + self.assertIsNone(first["error"]) + self.assertEqual(first["drift"], { + "plugin_installed": False, + "source_installed": False, + "source_plugin": False, + }) + self.assertFalse(first["installed"]["mcp"]) + self.assertTrue(first["installed_manifest_matches"]) + launcher = self.bin_dir / "squad" + self.assertTrue(os.access(launcher, os.X_OK)) + version = subprocess.run( + [str(launcher), "--version"], + check=True, + text=True, + capture_output=True, + env=self.environment, + ) + self.assertEqual(version.stdout.strip(), "squad 0.1.0") + self.assertEqual(version.stderr, "") + + release = Path(first["current_target"]) + self.assertTrue((release / "core/adapters/codex/adapter.json").is_file()) + self.assertTrue((release / "core/integrations/grok/registration.json").is_file()) + self.assertTrue((release / "venv/lib").is_dir()) + before_state = (self.install_root / "install-state.json").read_bytes() + before_launcher = launcher.read_bytes() + + second = self.install() + self.assertFalse(second["changed"]) + self.assertEqual(second["current_target"], first["current_target"]) + self.assertEqual((self.install_root / "install-state.json").read_bytes(), before_state) + self.assertEqual(launcher.read_bytes(), before_launcher) + self.assertEqual(len(list((self.install_root / "releases").iterdir())), 1) + + def test_composite_installer_succeeds_without_claude(self): + self.environment["PATH"] = "/usr/bin:/bin" + result = subprocess.run( + ["/bin/bash", str(COMPOSITE_INSTALLER), "--core-only", "--json"], + check=True, + text=True, + capture_output=True, + env=self.environment, + cwd=ROOT, + ) + report = json.loads(result.stdout) + self.assertTrue(report["changed"]) + self.assertIn("Claude plugin: skipped", result.stderr) + self.assertEqual( + subprocess.run( + [str(self.bin_dir / "squad"), "--version"], + check=True, + text=True, + capture_output=True, + env=self.environment, + ).stdout.strip(), + "squad 0.1.0", + ) + + def test_composite_claude_reinstall_uses_plugin_hooks_once(self): + fake_bin = self.root / "fake-bin" + fake_bin.mkdir() + fake_claude = fake_bin / "claude" + fake_claude.write_text("""#!/bin/sh +set -eu +printf '%s\\n' "$*" >> "$DEVSQUAD_FAKE_CLAUDE_LOG" +case "$*" in + "plugin marketplace list") + if [ -f "$DEVSQUAD_FAKE_CLAUDE_STATE/marketplace" ]; then echo devsquad-marketplace; fi ;; + "plugin marketplace add "*) touch "$DEVSQUAD_FAKE_CLAUDE_STATE/marketplace" ;; + "plugin marketplace update "*) : ;; + "plugin list") + if [ -f "$DEVSQUAD_FAKE_CLAUDE_STATE/plugin" ]; then echo devsquad@devsquad-marketplace; fi ;; + "plugin install "*) touch "$DEVSQUAD_FAKE_CLAUDE_STATE/plugin" ;; + "plugin update "*) : ;; + "plugin enable "*) : ;; + *) exit 64 ;; +esac +""") + fake_claude.chmod(0o755) + state = self.root / "fake-claude-state" + state.mkdir() + log = self.root / "fake-claude.log" + home = self.root / "home" + settings = home / ".claude/settings.json" + settings.parent.mkdir(parents=True) + original_settings = '{"hooks":{"sentinel":[]}}\n' + settings.write_text(original_settings) + environment = self.environment.copy() + environment.update({ + "PATH": f"{fake_bin}:/usr/bin:/bin", + "HOME": str(home), + "DEVSQUAD_FAKE_CLAUDE_LOG": str(log), + "DEVSQUAD_FAKE_CLAUDE_STATE": str(state), + }) + command = ["/bin/bash", str(COMPOSITE_INSTALLER), "--with-claude"] + first = subprocess.run( + command, text=True, capture_output=True, env=environment, cwd=ROOT, + ) + self.assertEqual(first.returncode, 0, (first.stdout, first.stderr)) + second = subprocess.run( + command, text=True, capture_output=True, env=environment, cwd=ROOT, + ) + self.assertEqual(second.returncode, 0, (second.stdout, second.stderr)) + self.assertIn("Claude plugin ready", first.stdout) + self.assertIn("Claude plugin ready", second.stdout) + self.assertEqual(settings.read_text(), original_settings) + calls = log.read_text().splitlines() + self.assertEqual(calls, [ + "plugin marketplace list", + "plugin marketplace add https://github.com/joshidikshant/devsquad.git", + "plugin list", + "plugin install devsquad@devsquad-marketplace", + "plugin enable devsquad@devsquad-marketplace", + "plugin marketplace list", + "plugin marketplace update devsquad-marketplace", + "plugin list", + "plugin update devsquad@devsquad-marketplace", + "plugin enable devsquad@devsquad-marketplace", + ]) + + def test_marketplace_package_source_contains_the_complete_core(self): + marketplace = json.loads((ROOT / ".claude-plugin/marketplace.json").read_text()) + entries = [item for item in marketplace["plugins"] if item["name"] == "devsquad"] + self.assertEqual(len(entries), 1) + plugin_root = (ROOT / entries[0]["source"]).resolve() + self.assertEqual(plugin_root, (ROOT / "plugin").resolve()) + required = { + "core/pyproject.toml", + "core/bin/squad", + "core/src/devsquad/cli.py", + "core/src/devsquad/migrations/013_decision_observations.sql", + "core/integrations/codex/registration.json", + "core/schemas/task.schema.json", + } + self.assertEqual( + {path for path in required if not (plugin_root / path).is_file()}, + set(), + ) + + def test_status_reports_drift_and_update_selects_a_new_immutable_release(self): + first = self.install() + old_release = Path(first["current_target"]) + old_manifest = (old_release / "release.json").read_bytes() + source = self.changed_source() + + status = self.install(source, "--status") + self.assertFalse(status["changed"]) + self.assertEqual(status["drift"], { + "plugin_installed": False, + "source_installed": True, + "source_plugin": True, + }) + + updated = self.install(source) + self.assertTrue(updated["changed"]) + self.assertNotEqual(updated["current_target"], first["current_target"]) + self.assertEqual(updated["installed"]["version"], "0.1.1") + self.assertEqual(updated["drift"], { + "plugin_installed": True, + "source_installed": False, + "source_plugin": True, + }) + self.assertTrue(old_release.is_dir()) + self.assertEqual((old_release / "release.json").read_bytes(), old_manifest) + self.assertEqual(len(list((self.install_root / "releases").iterdir())), 2) + version = subprocess.run( + [str(self.bin_dir / "squad"), "--version"], + check=True, + text=True, + capture_output=True, + env=self.environment, + ) + self.assertEqual(version.stdout.strip(), "squad 0.1.1") + + def test_update_does_not_break_an_active_release_pinned_run(self): + first = self.install() + old_release = Path(first["current_target"]) + old_python = old_release / "venv/bin/python" + repo = self.root / "repo" + runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + subprocess.run( + ["git", "-C", str(repo), "config", "user.email", "test@example.invalid"], + check=True, + ) + subprocess.run( + ["git", "-C", str(repo), "config", "user.name", "Test"], + check=True, + ) + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "src/app.py").write_text("VALUE = 'base'\n") + (repo / "tests/test_app.py").write_text("# fixture test\n") + profiles, policy = branch_review_routing_documents() + (repo / "profiles.json").write_text(profiles) + (repo / "policy.json").write_text(policy) + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(repo), "commit", "-qm", "base"], check=True) + task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + task["project"] = { + "repo_path": str(repo), "base_ref": "HEAD", "target_ref": "HEAD", + } + task["routing"] = { + "profiles_file": "profiles.json", "policy_file": "policy.json", + } + task_path = self.root / "active-task.json" + task_path.write_text(json.dumps(task)) + start_script = """ +import json +from pathlib import Path +import sys +from devsquad.service import Service +task = json.loads(Path(sys.argv[1]).read_text()) +print(json.dumps(Service(Path(sys.argv[2])).start( + task, "installer-update", _internal_fake_delay=4, +))) +""" + started_process = subprocess.run( + [str(old_python), "-P", "-c", start_script, str(task_path), str(runtime)], + check=True, + text=True, + capture_output=True, + env=self.environment, + cwd=self.root, + ) + started = json.loads(started_process.stdout) + self.assertEqual(started["state"], "queued", started) + + launcher = self.bin_dir / "squad" + deadline = time.monotonic() + 8 + before_update = None + while time.monotonic() < deadline: + before_update = self.cli_json(launcher, "status", started["run_id"], "--runtime-dir", str(runtime)) + if before_update["data"]["state"] == "running": + break + time.sleep(0.05) + self.assertEqual(before_update["data"]["state"], "running", before_update) + + updated = self.install(self.changed_source()) + self.assertNotEqual(updated["current_target"], str(old_release)) + self.assertTrue(old_release.is_dir()) + + deadline = time.monotonic() + 10 + after_update = None + while time.monotonic() < deadline: + after_update = self.cli_json(launcher, "status", started["run_id"], "--runtime-dir", str(runtime)) + if after_update["data"]["state"] == "succeeded": + break + time.sleep(0.05) + self.assertEqual(after_update["data"]["state"], "succeeded", after_update) + result = self.cli_json( + launcher, "result", started["run_id"], "--runtime-dir", str(runtime), + ) + self.assertTrue(result["data"]["ready"]) + with sqlite3.connect(runtime / "state.sqlite3") as connection: + package_path, package_digest = connection.execute( + "SELECT package_path, package_digest FROM runs WHERE id=?", + (started["run_id"],), + ).fetchone() + self.assertTrue(Path(package_path).is_dir()) + self.assertEqual(len(package_digest), 64) + + def cli_json(self, launcher, *arguments): + result = subprocess.run( + [str(launcher), *arguments, "--json"], + check=True, + text=True, + capture_output=True, + env=self.environment, + cwd=self.root, + ) + self.assertEqual(result.stderr, "") + return json.loads(result.stdout) + + +if __name__ == "__main__": + unittest.main() From d1b0c4c28aa7b3d8715ebca4253e7e54ccb731a7 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 03:24:32 -0700 Subject: [PATCH 117/197] docs: publish M7 runtime operations --- .github/workflows/offline.yml | 66 ++++++++ CONTRIBUTING.md | 10 ++ docs/ARCHITECTURE.md | 30 ++-- docs/RUNTIME-GUIDE.md | 154 ++++++++++++++++++ docs/generated/core-reference.md | 141 ++++++++++++++++ .../engineering-team/MCP-LOCAL-ACCESS.md | 10 +- .../plans/engineering-team/examples/README.md | 11 +- scripts/generate-core-reference.py | 127 +++++++++++++++ test/test_m7_packaging.sh | 67 ++++++++ 9 files changed, 601 insertions(+), 15 deletions(-) create mode 100644 .github/workflows/offline.yml create mode 100644 docs/RUNTIME-GUIDE.md create mode 100644 docs/generated/core-reference.md create mode 100755 scripts/generate-core-reference.py create mode 100755 test/test_m7_packaging.sh diff --git a/.github/workflows/offline.yml b/.github/workflows/offline.yml new file mode 100644 index 0000000..f1e41ad --- /dev/null +++ b/.github/workflows/offline.yml @@ -0,0 +1,66 @@ +name: Offline compatibility + +on: + push: + pull_request: + workflow_dispatch: + +permissions: + contents: read + +jobs: + legacy-bash: + name: Legacy Bash 3.2 (macOS) + runs-on: macos-latest + steps: + - uses: actions/checkout@v4 + - name: Verify system Bash floor + run: test "$(/bin/bash -c 'printf "%s.%s" "${BASH_VERSINFO[0]}" "${BASH_VERSINFO[1]}"')" = "3.2" + - name: Run legacy and packaging contracts + run: /bin/bash test/run.sh + + core: + name: Core Python ${{ matrix.python }} (macOS) + runs-on: macos-latest + strategy: + fail-fast: false + matrix: + python: ["3.11", "3.14"] + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python }} + cache: pip + - name: Install build-only test tooling + run: python -m pip install "setuptools>=68" wheel + - name: Verify generated command and schema reference + run: python scripts/generate-core-reference.py --check + - name: Run dependency-free core, package and native-protocol fixtures + env: + PYTHONDONTWRITEBYTECODE: "1" + PYTHONPATH: plugin/core/src + PYTHONWARNINGS: error::ResourceWarning + run: python -m unittest discover -s test/core -q + + optional-mcp: + name: Optional MCP 2.2.0 (provider-offline) + runs-on: macos-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.11" + cache: pip + - name: Install exact optional transport lock + run: python -m pip install -r plugin/core/requirements-mcp.lock + - name: Run MCP SDK and stdio compatibility fixtures + env: + PYTHONDONTWRITEBYTECODE: "1" + PYTHONPATH: plugin/core/src + PYTHONWARNINGS: error::ResourceWarning + run: python test/core/test_mcp.py -q + +# Test bodies make no provider or application network calls. Dependency setup +# resolves pinned public packages; live provider/app smoke runs remain explicit, +# bounded receipts outside CI. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index aac70ea..2cb4c0f 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -6,6 +6,9 @@ git clone https://github.com/joshidikshant/devsquad.git cd devsquad bash test/run.sh # no network, no real CLIs required — should be all green +PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src \ + python3 -m unittest discover -s test/core +python3 scripts/generate-core-reference.py --check ``` The test suite (`test/run.sh`) is the contract. It runs offline against fake @@ -54,3 +57,10 @@ Bump the version in `plugin/.claude-plugin/plugin.json` and both entries in `.claude-plugin/marketplace.json`, add a `CHANGELOG.md` entry, then after pushing run `claude plugin update devsquad@devsquad-marketplace`. Never point hook commands at a versioned cache dir — that freezes hooks at install time. + +Before a release, also run a fresh standalone install into temporary +`DEVSQUAD_INSTALL_ROOT` and `DEVSQUAD_BIN_DIR` locations. The tracked installer +tests cover Claude-free installation, idempotence, actual payload drift, +plugin contents and an active run surviving a release switch. The operator +commands and supported/deferred surface boundaries live in +[docs/RUNTIME-GUIDE.md](docs/RUNTIME-GUIDE.md). diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index ca84c5c..853e1ac 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -126,16 +126,26 @@ Enforced by `test/test_wrapper_contract.sh` (offline, fake CLI binaries): ## Deployment modes (the drift trap) -- **User mode**: `install.sh` registers global hooks pointing at the - marketplace clone (`~/.claude/plugins/marketplaces/devsquad-marketplace/plugin`). - Refresh with `claude plugin marketplace update devsquad-marketplace` + - `claude plugin update devsquad@devsquad-marketplace` after each release. -- **Dev mode** (this machine): global hooks point at the source checkout, so - hooks run at HEAD. Agents/commands/skills STILL load from the installed - plugin — after pushing, update the plugin or subagents run stale code. -- Never let hook commands reference a versioned cache dir - (`plugins/cache/...//`): that froze production at 0.3.0 for - five months while fixes accumulated unreleased. +- **Standalone mode**: `scripts/install-core.sh` creates an immutable release + below `~/.devsquad/releases/`, atomically selects it through + `~/.devsquad/current` and keeps `~/.local/bin/squad` stable. It needs no + Claude installation. `--status --json` compares source, plugin and actual + installed payload digests. +- **Claude plugin mode**: `install.sh` installs the standalone runtime first + and, when Claude is available, installs or updates the marketplace plugin. + `plugin/hooks/hooks.json` is the hook registration source. The installer no + longer writes a duplicate set into global Claude settings. +- **Dev mode**: run legacy hooks from this source checkout only when explicitly + testing changes at HEAD. Agents/commands/skills may still come from an + installed plugin, so doctor and install-status output must be checked before + attributing behavior to the source tree. +- Never point a stable launcher or manual hook at a versioned Claude cache + directory (`plugins/cache/...//`). Content-addressed standalone + releases are retained for active runs; the selector, not a running process, + moves during updates. + +See [the runtime guide](RUNTIME-GUIDE.md) for install, operation, recovery and +the honestly blocked live-surface matrix. ## State (per project, `.devsquad/`, self-gitignored) diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md new file mode 100644 index 0000000..31bb520 --- /dev/null +++ b/docs/RUNTIME-GUIDE.md @@ -0,0 +1,154 @@ +# DevSquad runtime guide + +This guide covers the surface-independent Python runtime. The older Claude +plugin remains available, but it is not required for the standalone command. + +## Install and update + +Prerequisites are macOS or another Unix-like host and Python 3.11 or newer. +The base runtime has no third-party Python dependencies and installation does +not contact a provider or package index. + +From a DevSquad checkout: + +```bash +./install.sh --core-only +export PATH="$HOME/.local/bin:$PATH" +squad --version +./scripts/install-core.sh --status --json +``` + +`scripts/install-core.sh` creates a Python virtual environment and an exact +copy of `plugin/core` under an immutable, content-addressed directory in +`~/.devsquad/releases/`. `~/.devsquad/current` is changed atomically and +`~/.local/bin/squad` remains stable. A second identical install reports +`"changed":false`. The status report compares the source, plugin and actual +installed payload digests; it does not trust the release manifest alone. + +`./install.sh` also installs or updates the legacy Claude plugin when the +`claude` command exists. Use `--with-claude` to require that path. The plugin's +`hooks/hooks.json` is the only hook registration written by the installer; +the composite installer does not add another copy to global settings. + +The optional local MCP bridge is isolated from the dependency-free base. It +uses the exact `mcp==2.2.0` lock and never downloads implicitly. Prepare a +wheelhouse containing every package in `plugin/core/requirements-mcp.lock`, +then run: + +```bash +./scripts/install-core.sh --with-mcp --mcp-wheelhouse /absolute/path/to/wheels +squad setup --dry-run --json +squad setup --json +squad doctor --json +``` + +Setup registers only one stable `squad` launcher. It refuses duplicate, +inherited, malformed or ambiguous MCP registrations instead of guessing which +one to replace. An upgrade retains every previous release, so a process that +started before the selector changed can finish against its frozen package. + +## Operate a run + +The generated [command and schema reference](generated/core-reference.md) +lists every command form, packaged schema digest and a strict task-shape +example. A real task must name an existing Git repository, committed refs and +committed routing files containing locally verified profiles. + +```bash +squad start --task-file task.json --idempotency-key issue-123 --json +squad status RUN_ID --json +squad events RUN_ID --after 0 --limit 100 --json +squad result RUN_ID --json +``` + +Keep the returned run ID. Reusing an idempotency key with an identical request +returns the original run; reusing it with different content conflicts. Status +is the authority for the current state and next action. Result artifacts and +their SHA-256 values are authoritative; a chat summary is not. + +When a host-lead workflow pauses, status returns `claim_handoff` and the +current run version. Claim and complete the saved packet without editing the +claim: + +```bash +squad handoff claim RUN_ID --expected-version VERSION --owner local-operator --json > claim-response.json +python3 -c 'import json,sys; json.dump(json.load(sys.stdin)["data"]["claim"],sys.stdout)' < claim-response.json > claim.json +squad handoff complete RUN_ID --claim-file claim.json --decision-file decision.json --json +``` + +The initial claim command returns the claim; `--claim-file` on that command is +only for renewing an existing claim. Construct `decision.json` from the exact +packet and evidence references returned with the claim, following the strict +[handoff contract](plans/engineering-team/CONTRACTS.md). + +Every surface uses these operations directly or through the thin local MCP +bridge described in [MCP-LOCAL-ACCESS.md](plans/engineering-team/MCP-LOCAL-ACCESS.md). +Closing an app does not cancel the detached run. + +## Recover or cancel + +Never delete the runtime database, an active release or a run-owned worktree +to recover a job. Inspect first: + +```bash +squad status RUN_ID --json +squad events RUN_ID --after 0 --limit 100 --json +``` + +If status supplies a recovery object, save that exact object as +`recovery.json`, reconcile the process identity it describes, and then run: + +```bash +squad resume RUN_ID --recovery-file recovery.json --json +``` + +If status does not request recovery, do not invent a recovery decision. +Cancellation is durable and idempotent: + +```bash +squad cancel RUN_ID --json +squad status RUN_ID --json +``` + +For installation drift, run `scripts/install-core.sh --status --json` from the +intended checkout. Reinstalling identical content is safe. A mismatched or +corrupt content-addressed release is rejected rather than repaired in place; +install a new source digest and retain the old directory for run evidence. + +## Supported and deferred boundaries + +The current packaged contract is Python 3.11+, public JSON contract version 1, +SQLite schema 13 and optional MCP SDK 2.2.0 exactly. Native Codex fixtures and +the recorded live proof cover bundled `codex-cli 0.153.4`. The Claude worker +adapter is version-scoped to CLI 2.1.220. Antigravity 1.2.3 and Grok 0.2.111 +have registration evidence but their live worker paths remain blocked by +trust/permission and expired authentication respectively. Version changes are +capability drift and require a fresh conformance probe; brand names are not a +compatibility promise. + +Implemented surfaces and evidence: + +| Surface | Current evidence | +|---|---| +| Terminal | Standalone install, start/status/result/cancel/recovery and update-survival tests | +| Codex App/CLI | Matching MCP registration and a real saved-run `squad_status` receipt | +| Claude Code local Code tab | Matching registration; live operation blocked on normal provider login | +| Antigravity local IDE/CLI | Matching registration; live scoped operation blocked on trust/permission | +| Grok Build | Matching registration; live operation blocked on expired authentication | + +The evidence source is +[`M4-local-mcp-2026-09-23.json`](plans/engineering-team/evidence/M4-local-mcp-2026-09-23.json). +M7 requires new installed-runtime receipts before claiming universal surface +support. + +Portable task files, handoff packets, event ledgers and hashed artifacts are +the cross-host interface. A reviewed upstream Codex integration demonstrates +optional native Claude-to-Codex transcript import, but DevSquad does not yet +expose or test that import path. It is deferred and must never be substituted +for the portable handoff contract or described as general chat-history +transfer. + +The optional Jev decision probe remains blocked until `TYPESAFE_API_KEY` is +provided; normal routing defaults to the deterministic zero-call path. Laya is +not installed unless the declared Jev fallback trigger fires. The optional C1 +Council extension is also deferred and does not block the core runtime. diff --git a/docs/generated/core-reference.md b/docs/generated/core-reference.md new file mode 100644 index 0000000..772230a --- /dev/null +++ b/docs/generated/core-reference.md @@ -0,0 +1,141 @@ + +# DevSquad core command and schema reference + +Regenerate with `python3 scripts/generate-core-reference.py`; verify with +`python3 scripts/generate-core-reference.py --check`. + +## Command forms + +- `squad cancel [-h] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad capacity [-h] {observe} ...` +- `squad capacity observe [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad classify [-h] [--cwd CWD] [--model MODEL] [--effort EFFORT] [--permission {read_only,workspace_write}] [--timeout TIMEOUT] [--transport {cli_exec,native_protocol}] [--catalog-file CATALOG_FILE] --returncode RETURNCODE --stdout-file STDOUT_FILE --stderr-file STDERR_FILE {codex,antigravity,grok}` +- `squad doctor [-h] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` +- `squad events [-h] [--after AFTER] [--limit LIMIT] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad handoff [-h] {claim,complete} ...` +- `squad handoff claim [-h] --expected-version EXPECTED_VERSION --owner OWNER [--claim-file CLAIM_FILE] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad handoff complete [-h] --claim-file CLAIM_FILE --decision-file DECISION_FILE [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad learn [-h] {propose} ...` +- `squad learn propose [-h] --project PROJECT [--json] [--runtime-dir RUNTIME_DIR]` +- `squad mcp [-h] {serve} ...` +- `squad mcp serve [-h] [--runtime-dir RUNTIME_DIR] [--surface SURFACE] [--session-ref SESSION_REF]` +- `squad outcome [-h] {add} ...` +- `squad outcome add [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad policy [-h] {evaluate} ...` +- `squad policy evaluate [-h] --experiment EXPERIMENT [--json] [--runtime-dir RUNTIME_DIR]` +- `squad prepare [-h] [--cwd CWD] [--model MODEL] [--effort EFFORT] [--permission {read_only,workspace_write}] [--timeout TIMEOUT] [--transport {cli_exec,native_protocol}] [--catalog-file CATALOG_FILE] --prompt PROMPT {codex,antigravity,grok}` +- `squad profile [-h] {template-add,binding-bootstrap,qualification-add,binding-change,binding-fallback,binding-show} ...` +- `squad profile binding-bootstrap [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad profile binding-change [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad profile binding-fallback [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad profile binding-show [-h] [--json] [--runtime-dir RUNTIME_DIR] alias` +- `squad profile qualification-add [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad profile template-add [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` +- `squad report [-h] --project PROJECT [--json] [--runtime-dir RUNTIME_DIR]` +- `squad result [-h] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad resume [-h] [--json] [--runtime-dir RUNTIME_DIR] [--recovery-file RECOVERY_FILE] run` +- `squad setup [-h] [--host {codex,claude-code,antigravity,grok}] [--dry-run] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` +- `squad start [-h] --task-file TASK_FILE --idempotency-key IDEMPOTENCY_KEY [--supersedes-run SUPERSEDES_RUN] [--wait] [--json] [--runtime-dir RUNTIME_DIR]` +- `squad status [-h] [--json] [--runtime-dir RUNTIME_DIR] run` + +## Common command examples + +```bash +squad --version +squad doctor --project-dir "$PWD" --json +squad start --task-file task.json --idempotency-key issue-123 --json +squad status RUN_ID --json +squad events RUN_ID --after 0 --limit 100 --json +squad result RUN_ID --json +squad cancel RUN_ID --json +squad resume RUN_ID --recovery-file recovery.json --json +squad setup --host codex --dry-run --json +``` + +`RUN_ID`, task paths and recovery evidence are operator-supplied values; the +runtime never infers them from chat history. + +## Packaged JSON schemas + +| File | Identifier | Required top-level fields | SHA-256 | +|---|---|---|---| +| `adapter.schema.json` | `https://devsquad.local/schemas/adapter-v1.json` | `schema_version`, `name`, `transport`, `binary_candidates`, `capabilities`, `permission_profiles` | `fc6b811b59bd2920d3dc1588d5bf447515a25d87219d07d76e1764efd60ad9c0` | +| `check-result.schema.json` | `https://devsquad.local/schemas/check-result-v1.json` | `schema_version`, `candidate_sha256`, `target_oid`, `id`, `argv`, `cwd`, `required_to_pass`, `status`, `returncode`, `error_code`, `duration_ms`, `stdout`, `stderr` | `6597b7e6839d6b215c1bb692c3ee5db47ee88495d96bdfa4730db6f56b2c0c32` | +| `execution-identity.schema.json` | `https://devsquad.local/schemas/execution-identity-v1.json` | `harness`, `harness_version`, `model_provider`, `model_family`, `model`, `effort`, `tools`, `permissions`, `account_pool`, `verification` | `99386fb81c3567bdd3c880b37d363d5ca61cccb3ff35add5eb01a1ad916dfa90` | +| `launch-spec.schema.json` | `https://devsquad.local/schemas/launch-spec-v1.json` | `schema_version`, `adapter`, `transport`, `argv`, `cwd`, `stdin_path`, `timeout_seconds`, `requested`, `environment` | `12e2c3ffbfee6b3a9e93e41d9dd7d6a2c82a9d62e981d9378307b12f6ff12f5c` | +| `normalized-result.schema.json` | `https://devsquad.local/schemas/normalized-result-v1.json` | `schema_version`, `execution_status`, `error_code`, `output`, `artifact_status`, `acceptance_status`, `requested`, `observed`, `native_ids`, `events` | `8fb7eec0d3cfc952b9fa839b5fdbe950d5cfe06ed69bbbd3bcf835aad9df753d` | +| `policy.schema.json` | `https://devsquad.local/schemas/policy-v1.json` | `schema_version`, `id`, `version`, `roles`, `task_classes`, `require_different_model_for_review`, `account_pools`, `experiment_budget` | `00009a5a1fdfe31b766561e4e890ba75fed25b9692f9732ebf676e63860da50c` | +| `profile.schema.json` | `https://devsquad.local/schemas/profile-v1.json` | `id`, `harness`, `model_family`, `model_id`, `effort`, `required_tools`, `permission_policy`, `account_pool_id`, `billing_mode`, `quality_status`, `evidence_refs` | `fc51b49cd7dd6a134eedf2c2c941301392aef29a2163ac2d307b4fb78509ec96` | +| `profiles.schema.json` | `https://devsquad.local/schemas/profiles-v1.json` | `schema_version`, `profiles`, `bindings` | `c14823c89d1b81ee93b74f5bbc2701ae975825db795b5f0937c28fc057d84cf1` | +| `review-result.schema.json` | `https://devsquad.local/schemas/review-result-v1.json` | `schema_version`, `candidate_sha256`, `base_oid`, `target_oid`, `review_mode`, `verdict`, `summary`, `findings` | `96b852543f92dca21dafd6c0bc954f8f56d3dbfb0981d601cc05b9c1c9ba6e6b` | +| `task.schema.json` | `https://devsquad.local/schemas/task-v1.json` | `schema_version`, `project`, `workflow`, `goal`, `task_class`, `acceptance`, `checks`, `scope`, `lead`, `routing`, `budget`, `origin` | `467fcb30221db38c842fcc9eb4c107f3cc90f390de332c1d50576e0db37afc9f` | + +## Task-shape example + +This generated copy demonstrates the strict v1 task shape. Replace the +fixture repository, refs, routing files and checks before starting a run. + +```json +{ + "acceptance": [ + { + "description": "Findings identify the resolved candidate and supporting file locations.", + "evidence_kind": "review", + "id": "review-exact-diff" + }, + { + "description": "Include the fixture check result even when it reports a failure.", + "evidence_kind": "check", + "id": "report-check-outcome" + } + ], + "budget": { + "max_fallbacks_per_step": 1, + "max_revisions": 0, + "max_worker_invocations": 3, + "wall_seconds": 600 + }, + "checks": [ + { + "argv": [ + "python3", + "-m", + "unittest", + "discover", + "-s", + "tests" + ], + "cwd": ".", + "id": "fixture-tests", + "required_to_pass": false, + "timeout_seconds": 30 + } + ], + "goal": "Review the fixture change and report supported correctness findings.", + "lead": { + "mode": "host" + }, + "origin": { + "surface": "cli" + }, + "project": { + "base_ref": "fixture-base", + "repo_path": "/absolute/path/to/fixture-repository", + "target_ref": "fixture-candidate" + }, + "routing": { + "policy_file": "devsquad/policy.json", + "profiles_file": "devsquad/profiles.json" + }, + "schema_version": 1, + "scope": { + "read_paths": [ + "src", + "tests" + ], + "write_paths": [] + }, + "task_class": "fixture-review-small", + "workflow": "branch-review" +} +``` diff --git a/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md b/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md index 6662784..ea9fea7 100644 --- a/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md +++ b/docs/plans/engineering-team/MCP-LOCAL-ACCESS.md @@ -50,11 +50,13 @@ the service checks its fencing token and run version. | Antigravity | `agy mcp add` | Antigravity user config | `agy mcp list` | | Grok Build | `grok mcp add --scope user` | user | `grok mcp list --json` | -Install the optional, exactly pinned MCP extra into the same stable environment -that owns the `squad` executable, then preview or apply registration: +Install the optional, exactly pinned MCP extra into the same immutable release +that owns the `squad` executable. The installer is offline-only and requires a +wheelhouse containing every locked dependency; then preview or apply +registration: ```text -python3 -m pip install './plugin/core[mcp]' +./scripts/install-core.sh --with-mcp --mcp-wheelhouse /absolute/path/to/wheels squad setup --dry-run --json squad setup --json squad doctor --json @@ -89,3 +91,5 @@ installed app is not ready, while unavailable apps are not treated as required. These commands prove local CLI registration and SDK conformance. Actual in-app operation still requires the cross-surface receipts in M4/M7; config syntax or a matching `mcp list` result is not presented as that live proof. +The full install, operate and recovery procedure is in +[the runtime guide](../../RUNTIME-GUIDE.md). diff --git a/docs/plans/engineering-team/examples/README.md b/docs/plans/engineering-team/examples/README.md index a716462..139e4ab 100644 --- a/docs/plans/engineering-team/examples/README.md +++ b/docs/plans/engineering-team/examples/README.md @@ -1,6 +1,9 @@ # Design fixtures -These task files show the v1 shape. They are **not runnable against today's DevSquad**. Repository paths, refs, task classes and test commands refer to a future disposable fixture repository. M1 creates the schemas; M3/M5 create the corresponding repositories and executable tests. +These task files show the implemented strict v1 shape. They are not directly +runnable as checked in: their repository paths, refs, routing files and test +commands deliberately name disposable fixtures. M3 and M5 exercise equivalent +tasks end to end against temporary Git repositories. - [branch-review.json](branch-review.json): a host lead receives the review packet; a failing report-only check becomes a finding. - [issue-delivery.json](issue-delivery.json): a headless lead resolves a bounded implementation/review workflow; required tests must pass. @@ -14,4 +17,8 @@ configuration uses exact locally verified models and effort settings. These examples intentionally avoid embedding today's provider model names or pretending that fixture profiles are proven. -The CLI and MCP service must validate equivalent task objects against the same generated schema. Validate paths/refs against the temporary repository only in integration tests; ordinary schema tests validate shape without touching the filesystem. +The CLI and MCP service validate equivalent task objects against the same +packaged contract. The generated command/schema reference is +[docs/generated/core-reference.md](../../../generated/core-reference.md). +Integration tests resolve paths and refs against temporary repositories; +ordinary schema tests validate shape without touching the filesystem. diff --git a/scripts/generate-core-reference.py b/scripts/generate-core-reference.py new file mode 100755 index 0000000..5a99ada --- /dev/null +++ b/scripts/generate-core-reference.py @@ -0,0 +1,127 @@ +#!/usr/bin/env python3 +"""Generate the checked-in DevSquad CLI and schema reference.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +from pathlib import Path +import sys + + +ROOT = Path(__file__).resolve().parents[1] +CORE = ROOT / "plugin/core" +OUTPUT = ROOT / "docs/generated/core-reference.md" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.cli import parser as build_parser # noqa: E402 + + +def command_usages() -> list[str]: + result: list[str] = [] + + def visit(command_parser: argparse.ArgumentParser) -> None: + for action in command_parser._actions: + if not isinstance(action, argparse._SubParsersAction): + continue + for name in sorted(action.choices): + child = action.choices[name] + usage = child.format_usage().strip().removeprefix("usage: ") + result.append(" ".join(usage.split())) + visit(child) + + visit(build_parser()) + return result + + +def schema_rows() -> list[tuple[str, str, str, str]]: + rows = [] + for path in sorted((CORE / "schemas").glob("*.schema.json")): + raw = path.read_bytes() + value = json.loads(raw) + required = value.get("required", []) + rows.append(( + path.name, + value.get("$id", "—"), + ", ".join(f"`{field}`" for field in required) or "—", + hashlib.sha256(raw).hexdigest(), + )) + return rows + + +def render() -> str: + task_example = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + lines = [ + "", + "# DevSquad core command and schema reference", + "", + "Regenerate with `python3 scripts/generate-core-reference.py`; verify with", + "`python3 scripts/generate-core-reference.py --check`.", + "", + "## Command forms", + "", + ] + lines.extend(f"- `{usage}`" for usage in command_usages()) + lines.extend([ + "", + "## Common command examples", + "", + "```bash", + "squad --version", + "squad doctor --project-dir \"$PWD\" --json", + "squad start --task-file task.json --idempotency-key issue-123 --json", + "squad status RUN_ID --json", + "squad events RUN_ID --after 0 --limit 100 --json", + "squad result RUN_ID --json", + "squad cancel RUN_ID --json", + "squad resume RUN_ID --recovery-file recovery.json --json", + "squad setup --host codex --dry-run --json", + "```", + "", + "`RUN_ID`, task paths and recovery evidence are operator-supplied values; the", + "runtime never infers them from chat history.", + "", + "## Packaged JSON schemas", + "", + "| File | Identifier | Required top-level fields | SHA-256 |", + "|---|---|---|---|", + ]) + for name, identifier, required, digest in schema_rows(): + lines.append(f"| `{name}` | `{identifier}` | {required} | `{digest}` |") + lines.extend([ + "", + "## Task-shape example", + "", + "This generated copy demonstrates the strict v1 task shape. Replace the", + "fixture repository, refs, routing files and checks before starting a run.", + "", + "```json", + json.dumps(task_example, indent=2, sort_keys=True), + "```", + "", + ]) + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + arguments = argparse.ArgumentParser() + arguments.add_argument("--check", action="store_true") + parsed = arguments.parse_args(argv) + content = render() + if parsed.check: + if not OUTPUT.is_file() or OUTPUT.read_text() != content: + print(f"generated reference is stale: {OUTPUT}", file=sys.stderr) + return 1 + print(f"generated reference is current: {OUTPUT}") + return 0 + OUTPUT.parent.mkdir(parents=True, exist_ok=True) + OUTPUT.write_text(content) + print(OUTPUT) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/test/test_m7_packaging.sh b/test/test_m7_packaging.sh new file mode 100755 index 0000000..dd84f7b --- /dev/null +++ b/test/test_m7_packaging.sh @@ -0,0 +1,67 @@ +#!/usr/bin/env bash +# M7 documentation, generator and installer compatibility checks. +# Bash 3.2 compatible; no network or provider CLIs. +set -u + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P)" +REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)" +PASS=0 +FAIL=0 + +pass() { PASS=$((PASS + 1)); } +fail() { FAIL=$((FAIL + 1)); echo " FAIL: $1"; } + +if /bin/bash -n "$REPO_ROOT/install.sh" "$REPO_ROOT/scripts/install-core.sh"; then + pass +else + fail "install scripts must parse under the system Bash" +fi + +if /bin/bash "$REPO_ROOT/install.sh" --help 2>/dev/null | grep -q -- '--core-only'; then + pass +else + fail "composite installer help must document standalone mode" +fi + +if /bin/bash "$REPO_ROOT/scripts/install-core.sh" --help 2>/dev/null | grep -q -- '--mcp-wheelhouse'; then + pass +else + fail "core installer help must document offline MCP installation" +fi + +if python3 "$REPO_ROOT/scripts/generate-core-reference.py" --check >/dev/null 2>&1; then + pass +else + fail "generated command/schema reference must be current" +fi + +if grep -q 'scripts/install-core.sh --status --json' "$REPO_ROOT/docs/RUNTIME-GUIDE.md"; then + pass +else + fail "runtime guide must document drift inspection" +fi + +if grep -q 'native Claude-to-Codex transcript import' "$REPO_ROOT/docs/RUNTIME-GUIDE.md"; then + pass +else + fail "runtime guide must state the native transcript-import boundary" +fi + +if python3 - "$REPO_ROOT/plugin/core/schemas" <<'PY' >/dev/null 2>&1 +import json +from pathlib import Path +import sys +paths = sorted(Path(sys.argv[1]).glob("*.schema.json")) +assert paths +for path in paths: + value = json.loads(path.read_text()) + assert value["$id"].startswith("https://devsquad.local/schemas/") +PY +then + pass +else + fail "all packaged schemas must parse and retain stable identifiers" +fi + +echo " m7_packaging: ${PASS} passed, ${FAIL} failed" +[ "$FAIL" -eq 0 ] From 258abbdab02863c80129611e8248da27848e02d5 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 03:26:34 -0700 Subject: [PATCH 118/197] fix: migrate legacy standalone launcher --- scripts/install-core.sh | 22 +++++++++++++++++-- test/core/test_install_core.py | 39 ++++++++++++++++++++++++++++++++++ 2 files changed, 59 insertions(+), 2 deletions(-) diff --git a/scripts/install-core.sh b/scripts/install-core.sh index 8603438..76892f9 100755 --- a/scripts/install-core.sh +++ b/scripts/install-core.sh @@ -412,8 +412,26 @@ PY fi LAUNCHER="$BIN_DIR/squad" -if [ -e "$LAUNCHER" ] && ! grep -q '^# managed-by: devsquad-install-core-v1$' "$LAUNCHER" 2>/dev/null; then - fail "refusing to replace unmanaged launcher: $LAUNCHER" +if { [ -e "$LAUNCHER" ] || [ -L "$LAUNCHER" ]; } \ + && ! grep -q '^# managed-by: devsquad-install-core-v1$' "$LAUNCHER" 2>/dev/null; then + if [ -L "$LAUNCHER" ]; then + "$PYTHON" - "$LAUNCHER" "$INSTALL_ROOT/releases" <<'PY' || fail "refusing to replace unmanaged launcher: $LAUNCHER" +from pathlib import Path +import sys + +launcher = Path(sys.argv[1]) +releases = Path(sys.argv[2]).resolve(strict=True) +target = launcher.resolve(strict=True) +try: + relative = target.relative_to(releases) +except ValueError as exc: + raise SystemExit("legacy launcher target is outside the release root") from exc +if len(relative.parts) != 4 or relative.parts[-2:] != ("bin", "squad") or relative.parts[-3] != "venv": + raise SystemExit("legacy launcher target is not a DevSquad release command") +PY + else + fail "refusing to replace unmanaged launcher: $LAUNCHER" + fi fi LAUNCHER_TEMP="$BIN_DIR/.squad.$$" "$PYTHON" - "$LAUNCHER_TEMP" "$INSTALL_ROOT" <<'PY' diff --git a/test/core/test_install_core.py b/test/core/test_install_core.py index 18f2391..03b5219 100644 --- a/test/core/test_install_core.py +++ b/test/core/test_install_core.py @@ -135,6 +135,45 @@ def test_composite_installer_succeeds_without_claude(self): "squad 0.1.0", ) + def test_known_legacy_release_symlink_is_migrated_but_arbitrary_launcher_is_not(self): + legacy = self.install_root / "releases/0.1.0+legacy/venv/bin" + legacy.mkdir(parents=True) + legacy_squad = legacy / "squad" + legacy_squad.write_text("#!/bin/sh\nexit 0\n") + legacy_squad.chmod(0o755) + self.bin_dir.mkdir() + launcher = self.bin_dir / "squad" + launcher.symlink_to(legacy_squad) + + migrated = self.install() + self.assertTrue(migrated["changed"]) + self.assertFalse(launcher.is_symlink()) + self.assertIn("managed-by: devsquad-install-core-v1", launcher.read_text()) + self.assertTrue(legacy_squad.is_file()) + + other_root = self.root / "other-install" + other_bin = self.root / "other-bin" + other_bin.mkdir() + arbitrary = self.root / "arbitrary-squad" + arbitrary.write_text("#!/bin/sh\nexit 0\n") + arbitrary.chmod(0o755) + (other_bin / "squad").symlink_to(arbitrary) + environment = { + **self.environment, + "DEVSQUAD_INSTALL_ROOT": str(other_root), + "DEVSQUAD_BIN_DIR": str(other_bin), + } + refused = subprocess.run( + ["/bin/bash", str(INSTALLER), "--json"], + text=True, + capture_output=True, + env=environment, + cwd=ROOT, + ) + self.assertNotEqual(refused.returncode, 0) + self.assertIn("refusing to replace unmanaged launcher", refused.stderr) + self.assertTrue((other_bin / "squad").is_symlink()) + def test_composite_claude_reinstall_uses_plugin_hooks_once(self): fake_bin = self.root / "fake-bin" fake_bin.mkdir() From 2275b49fdaef82245760054d3b149dc9a14cfa39 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 03:38:17 -0700 Subject: [PATCH 119/197] fix: validate offline MCP installations --- scripts/install-core.sh | 3 +- test/core/test_install_core.py | 54 ++++++++++++++++++++++++++++++++++ 2 files changed, 56 insertions(+), 1 deletion(-) diff --git a/scripts/install-core.sh b/scripts/install-core.sh index 76892f9..53643b9 100755 --- a/scripts/install-core.sh +++ b/scripts/install-core.sh @@ -353,9 +353,10 @@ dist.mkdir() (dist / "RECORD").write_text("") PY if [ "$WITH_MCP" -eq 1 ]; then - "$VENV_PYTHON" -m pip install --disable-pip-version-check --no-index \ + "$VENV_PYTHON" -m pip install --quiet --disable-pip-version-check --no-index \ --only-binary=:all: --find-links "$MCP_WHEELHOUSE" \ -r "$SOURCE_CORE/requirements-mcp.lock" + "$VENV_PYTHON" -m pip check >/dev/null "$VENV_PYTHON" - <<'PY' from importlib import metadata import mcp diff --git a/test/core/test_install_core.py b/test/core/test_install_core.py index 03b5219..e7f1cb3 100644 --- a/test/core/test_install_core.py +++ b/test/core/test_install_core.py @@ -1,6 +1,7 @@ import json import os from pathlib import Path +import re import shutil import sqlite3 import subprocess @@ -8,6 +9,7 @@ import tempfile import time import unittest +import zipfile from devsquad_test_fixtures import branch_review_routing_documents @@ -72,6 +74,37 @@ def changed_source(self): ) return destination + def mcp_wheelhouse(self): + wheelhouse = self.root / "mcp-wheelhouse" + wheelhouse.mkdir() + lock = CORE / "requirements-mcp.lock" + for raw_line in lock.read_text().splitlines(): + line = raw_line.strip() + if not line or line.startswith("#"): + continue + name, version = line.split("==", 1) + normalized = re.sub(r"[-_.]+", "_", name) + dist_info = f"{normalized}-{version}.dist-info" + wheel = wheelhouse / f"{normalized}-{version}-py3-none-any.whl" + with zipfile.ZipFile(wheel, "w", zipfile.ZIP_DEFLATED) as archive: + archive.writestr( + f"{dist_info}/METADATA", + "Metadata-Version: 2.1\n" + f"Name: {name}\n" + f"Version: {version}\n", + ) + archive.writestr( + f"{dist_info}/WHEEL", + "Wheel-Version: 1.0\n" + "Generator: devsquad-test\n" + "Root-Is-Purelib: true\n" + "Tag: py3-none-any\n", + ) + archive.writestr(f"{dist_info}/RECORD", "") + if name == "mcp": + archive.writestr("mcp/__init__.py", "") + return wheelhouse + def test_fresh_install_requires_no_claude_and_reinstall_is_idempotent(self): self.environment["PATH"] = "/usr/bin:/bin" @@ -111,6 +144,27 @@ def test_fresh_install_requires_no_claude_and_reinstall_is_idempotent(self): self.assertEqual(launcher.read_bytes(), before_launcher) self.assertEqual(len(list((self.install_root / "releases").iterdir())), 1) + def test_offline_mcp_install_keeps_json_clean_and_passes_dependency_check(self): + report = self.install( + CORE, + "--with-mcp", + "--mcp-wheelhouse", + str(self.mcp_wheelhouse()), + ) + + self.assertTrue(report["changed"]) + self.assertTrue(report["installed"]["mcp"]) + release = Path(report["current_target"]) + checked = subprocess.run( + [str(release / "venv/bin/python"), "-m", "pip", "check"], + check=True, + text=True, + capture_output=True, + env=self.environment, + ) + self.assertEqual(checked.stdout.strip(), "No broken requirements found.") + self.assertEqual(checked.stderr, "") + def test_composite_installer_succeeds_without_claude(self): self.environment["PATH"] = "/usr/bin:/bin" result = subprocess.run( From bf26455dc5739e5ff409fde783eccbc91e9f476c Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 03:47:13 -0700 Subject: [PATCH 120/197] test: verify current Codex native protocol --- plugin/core/adapters/codex/adapter.json | 2 +- test/core/test_m1.py | 2 ++ 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/plugin/core/adapters/codex/adapter.json b/plugin/core/adapters/codex/adapter.json index 67fb311..35b6e19 100644 --- a/plugin/core/adapters/codex/adapter.json +++ b/plugin/core/adapters/codex/adapter.json @@ -5,7 +5,7 @@ "fallback_transport": "cli_exec", "binary_candidates": ["codex"], "model_provider": "openai", - "verified_harness_versions": ["codex-cli 0.153.4"], + "verified_harness_versions": ["codex-cli 0.153.4", "codex-cli 0.155.0-alpha.9.2"], "capabilities": {"efforts_by_model": {}, "native_model_list": true, "resume": true}, "permission_profiles": {"read_only": [], "workspace_write": []}, "output_format": "jsonl" diff --git a/test/core/test_m1.py b/test/core/test_m1.py index c1bfdc3..b29c169 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -99,6 +99,8 @@ def test_native_launch_is_preparation_only_and_version_scoped(self): self.assertEqual(spec.argv[-2:], ("--listen", "stdio://")) self.assertIn('model_reasoning_effort="low"', spec.argv) self.assertEqual(spec.environment["DEVSQUAD_WORKER"], "1") + current = prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli 0.155.0-alpha.9.2") + self.assertEqual(current.transport, "native_protocol") with self.assertRaises(ContractError): prepare_native_codex(manifest, cwd=temp.name, model="gpt-test", effort="low", permission="read_only", timeout_seconds=9, harness_version_value="codex-cli future") From ccba125e342041ff2ef72fd7e0c9a9b44c356e7d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 03:51:20 -0700 Subject: [PATCH 121/197] fix: prefer verified Codex binary --- plugin/core/adapters/codex/adapter.json | 2 +- plugin/core/src/devsquad/adapters.py | 20 +++++++++++++++++- test/core/test_m1.py | 27 ++++++++++++++++++++++++- 3 files changed, 46 insertions(+), 3 deletions(-) diff --git a/plugin/core/adapters/codex/adapter.json b/plugin/core/adapters/codex/adapter.json index 35b6e19..9fdc58a 100644 --- a/plugin/core/adapters/codex/adapter.json +++ b/plugin/core/adapters/codex/adapter.json @@ -3,7 +3,7 @@ "name": "codex", "transport": "native_protocol", "fallback_transport": "cli_exec", - "binary_candidates": ["codex"], + "binary_candidates": ["codex", "/Applications/ChatGPT.app/Contents/Resources/codex", "/Applications/Codex.app/Contents/Resources/codex"], "model_provider": "openai", "verified_harness_versions": ["codex-cli 0.153.4", "codex-cli 0.155.0-alpha.9.2"], "capabilities": {"efforts_by_model": {}, "native_model_list": true, "resume": true}, diff --git a/plugin/core/src/devsquad/adapters.py b/plugin/core/src/devsquad/adapters.py index f87171d..ace6d04 100644 --- a/plugin/core/src/devsquad/adapters.py +++ b/plugin/core/src/devsquad/adapters.py @@ -64,7 +64,25 @@ def load(cls, path: Path) -> "AdapterManifest": ) def resolve_binary(self) -> str | None: - return next((p for name in self.binary_candidates if (p := shutil.which(name))), None) + available = tuple( + dict.fromkeys( + path + for name in self.binary_candidates + if (path := shutil.which(name)) is not None + ) + ) + if self.verified_versions: + verified = next( + ( + path + for path in available + if harness_version(path) in self.verified_versions + ), + None, + ) + if verified is not None: + return verified + return available[0] if available else None def with_model_efforts(self, mapping: dict[str, tuple[str, ...]]) -> "AdapterManifest": return AdapterManifest( diff --git a/test/core/test_m1.py b/test/core/test_m1.py index b29c169..fed27db 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -6,6 +6,7 @@ import tempfile import subprocess import unittest +from dataclasses import replace from pathlib import Path from unittest.mock import patch @@ -27,10 +28,34 @@ def manifest(self, name: str) -> AdapterManifest: def fake_path(self, name: str) -> tuple[tempfile.TemporaryDirectory, Path]: temp = tempfile.TemporaryDirectory() path = Path(temp.name) / name - path.write_text("#!/bin/sh\nexit 0\n") + if name == "codex": + path.write_text( + "#!/bin/sh\n" + "if [ \"${1:-}\" = --version ]; then\n" + " echo 'codex-cli 0.153.4'\n" + "fi\n" + "exit 0\n" + ) + else: + path.write_text("#!/bin/sh\nexit 0\n") path.chmod(path.stat().st_mode | stat.S_IXUSR) return temp, path + def test_verified_binary_candidate_wins_over_an_older_path_binary(self): + temp, older = self.fake_path("codex") + self.addCleanup(temp.cleanup) + older.write_text("#!/bin/sh\necho 'codex-cli 0.135.0'\n") + verified = Path(temp.name) / "bundled-codex" + verified.write_text("#!/bin/sh\necho 'codex-cli 0.155.0-alpha.9.2'\n") + verified.chmod(verified.stat().st_mode | stat.S_IXUSR) + manifest = replace( + self.manifest("codex"), + binary_candidates=("codex", str(verified)), + ) + + with patch.dict(os.environ, {"PATH": str(older.parent)}): + self.assertEqual(manifest.resolve_binary(), str(verified)) + def test_prepare_preserves_spaces_and_tsx_prompt(self): temp, binary = self.fake_path("agy") self.addCleanup(temp.cleanup) From 251ef70cdda4f874bf168a527f1829a5935873ae Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 03:57:40 -0700 Subject: [PATCH 122/197] docs: checkpoint installed M7 runtime --- docs/RUNTIME-GUIDE.md | 41 +++++-- docs/plans/engineering-team/M7-STATUS.md | 76 ++++++++++++ docs/plans/engineering-team/RESUME.md | 76 +++++++----- docs/plans/engineering-team/backlog.json | 27 +++- .../M7-installed-runtime-2026-09-29.json | 116 ++++++++++++++++++ 5 files changed, 290 insertions(+), 46 deletions(-) create mode 100644 docs/plans/engineering-team/M7-STATUS.md create mode 100644 docs/plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index 31bb520..50d4b06 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -47,6 +47,19 @@ inherited, malformed or ambiguous MCP registrations instead of guessing which one to replace. An upgrade retains every previous release, so a process that started before the selector changed can finish against its frozen package. +Antigravity's non-interactive print mode also enforces project permissions. +For unattended read-only status checks, add this exact grant to the DevSquad +project's Permissions list in Antigravity: + +```text +mcp(devsquad/squad_status) +``` + +This is narrower than a server wildcard and does not authorize terminal or +file access. Interactive use may instead approve the requested MCP operation +when prompted. Do not use the blanket permission-bypass option for setup or a +smoke test. + ## Operate a run The generated [command and schema reference](generated/core-reference.md) @@ -119,27 +132,29 @@ install a new source digest and retain the old directory for run evidence. The current packaged contract is Python 3.11+, public JSON contract version 1, SQLite schema 13 and optional MCP SDK 2.2.0 exactly. Native Codex fixtures and -the recorded live proof cover bundled `codex-cli 0.153.4`. The Claude worker -adapter is version-scoped to CLI 2.1.220. Antigravity 1.2.3 and Grok 0.2.111 -have registration evidence but their live worker paths remain blocked by -trust/permission and expired authentication respectively. Version changes are -capability drift and require a fresh conformance probe; brand names are not a -compatibility promise. +recorded live proofs cover bundled `codex-cli 0.153.4` and +`codex-cli 0.155.0-alpha.9.2`; the latter passed a fresh native initialize and +complete model-catalog probe. The resolver prefers that verified bundled +binary over an older unverified PATH binary. The Claude worker adapter is +version-scoped to CLI 2.1.220. Antigravity 1.2.13 has a live read-only MCP +receipt; Grok 0.2.111 has matching registration but expired authentication. +Version changes are capability drift and require a fresh conformance probe; +brand names are not a compatibility promise. Implemented surfaces and evidence: | Surface | Current evidence | |---|---| -| Terminal | Standalone install, start/status/result/cancel/recovery and update-survival tests | -| Codex App/CLI | Matching MCP registration and a real saved-run `squad_status` receipt | +| Terminal | Standalone install plus a real saved-run start and terminal cancellation | +| Codex App/CLI | Matching MCP registration and a fresh installed-runtime `squad_status` receipt on 0.155.0-alpha.9.2 | | Claude Code local Code tab | Matching registration; live operation blocked on normal provider login | -| Antigravity local IDE/CLI | Matching registration; live scoped operation blocked on trust/permission | +| Antigravity local IDE/CLI | Matching registration and a live Gemini `squad_status` receipt with one project-scoped grant | | Grok Build | Matching registration; live operation blocked on expired authentication | -The evidence source is -[`M4-local-mcp-2026-09-23.json`](plans/engineering-team/evidence/M4-local-mcp-2026-09-23.json). -M7 requires new installed-runtime receipts before claiming universal surface -support. +The current evidence source is +[`M7-installed-runtime-2026-09-29.json`](plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json). +Claude, Grok and the installed two-model delivery remain open, so universal +surface support is not yet claimed. Portable task files, handoff packets, event ledgers and hashed artifacts are the cross-host interface. A reviewed upstream Codex integration demonstrates diff --git a/docs/plans/engineering-team/M7-STATUS.md b/docs/plans/engineering-team/M7-STATUS.md new file mode 100644 index 0000000..d7e28a5 --- /dev/null +++ b/docs/plans/engineering-team/M7-STATUS.md @@ -0,0 +1,76 @@ +# M7 implementation status + +M7 is **in progress with its independent packaging work complete**. The +standalone runtime is installed and usable from terminal, Codex and +Antigravity. Universal surface completion remains blocked by normal Claude and +Grok authentication, and the installed different-model delivery is the same +external gate retained by M5. + +| Requirement | Evidence | Status | +|---|---|---| +| Fresh standalone install without Claude | Immutable base install and composite `--core-only` regressions | verified | +| Optional MCP environment | Exact offline lock, one-line JSON, `pip check`, official SDK tests | verified | +| Reinstall/update safety | Idempotence, drift detection, stable launcher, retained old release and active-run survival | verified | +| Legacy plugin migration | Complete `plugin/core` package, no duplicate global hooks, second setup unchanged | verified | +| Host registration | Four installed hosts load the stable launcher and report matching | verified | +| Terminal operation | Fresh saved run started and later cancelled at version 20 | verified live | +| Codex App/CLI operation | Ephemeral gpt-6-luna/low called installed `squad_status` on the saved run | verified live | +| Antigravity operation | Gemini 3.8 Flash Low called installed `squad_status` with one project-scoped grant | verified live | +| Claude Code operation | Registration matching; real operation requires normal login | blocked externally | +| Grok Build operation | Registration matching; real operation requires renewed authentication | blocked externally | +| Installed two-model delivery | Genuine Claude implementation followed by different-model Codex review/check/disposition | blocked with M5 | +| Documentation/CI | Runtime guide, generated command reference and macOS/Python/optional-MCP workflow | clean-home install/setup/doctor passed | +| Normal task entry | `squad review --base ...` and `squad fix "..."` without hand-written task JSON | not implemented; active M7 work | + +## Current installation + +The stable launcher is `/Users/Dikshant/.local/bin/squad`. The active immutable +release is +`0.1.0-py31214-4c5f034c67bf-mcp-a26bc88afbef`, using Python 3.12.14 and +`mcp==2.2.0`. A repeated install reports `changed=false`, `pip check` passes, +and `squad setup` reports every host unchanged. + +Codex capability drift was revalidated rather than inferred. Bundled +`codex-cli 0.155.0-alpha.9.2` passed `initialize` and a complete seven-model +`model/list`; the runtime now chooses that verified binary over PATH Codex +0.135.0. Unknown versions remain fail-closed. + +## Test gate + +- 300 Python core tests passed with `ResourceWarning` promoted to error; two + optional-SDK tests skipped in the dependency-free interpreter. +- 227/227 Bash assertions passed across 11 files. +- Eight focused installer tests passed. +- Twenty-two focused tests passed in the installed official MCP environment. +- The installed runtime passed `pip check`; setup and doctor report ready. + +The recurring interpreter-finalization `ResourceWarning` was printed as an +unraisable cleanup diagnostic during full discovery, but the warning-as-error +process completed successfully. It is retained as an observation, not +misreported as a failing test. + +## Live receipts + +The terminal-created run `f121e96b-502a-4986-a3c3-7ba8ab418bb3` was observed +through both Antigravity and Codex as `awaiting_host`, version 14, with next +action `claim_handoff`. Terminal cancellation then moved the same run and its +open handoff to `cancelled`, version 20. Raw provider output remains private; +hashes and redacted usage are in +[the portable M7 evidence](evidence/M7-installed-runtime-2026-09-29.json). + +Antigravity headless mode requires an explicit project grant for unattended +inspection. The verified minimum is `mcp(devsquad/squad_status)`; no global or +blanket permission bypass was used. Codex used an ephemeral minimal config, +read-only sandbox, no conversation resume and only the DevSquad MCP server. + +## Exact remaining work + +1. Implement and test the required `squad review` and `squad fix` task-entry + layer over the existing strict saved-run contracts. Keep `squad council` + attached to the separately gated C1 implementation. +2. After normal Claude login, execute the saved M4 handoff and the M5 genuine + Claude implementation to different-model Codex delivery. +3. After Grok login renewal, record one supported Grok Build operation against + the same installed runtime. +4. Re-run the final gates and mark M7 complete only when every required surface + receipt and installed delivery is present. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 57f86c0..50f059e 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -63,21 +63,20 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne a normally authenticated Claude Code host to perform the real handoff; the labeled SDK client is deliberately not presented as that proof. See the [portable redacted evidence](evidence/M4-local-mcp-2026-09-23.json). -- Last-observed provider readiness outside accepted M3: Claude CLI is not logged in; - Grok CLI authentication expired; Gemini CLI's individual-account path is - unsupported and its supported successor is Antigravity; Antigravity is - authenticated but headless execution still lacks scoped permission/trust. - Do not retry these blocked paths, buy credits, use paid API fallback or - change global provider settings. Resume a provider gate only after the user - completes the corresponding normal login/permission action. +- Last-observed provider readiness outside accepted M3: Claude CLI is not + logged in and Grok CLI authentication is expired. Antigravity 1.2.13 is + authenticated and its project-scoped `mcp(devsquad/squad_status)` grant has + now produced a live Gemini receipt. Do not retry the blocked Claude/Grok + paths, buy credits, use paid API fallback or change global provider settings. + Resume those gates only after the user completes the corresponding login. - User wants the implementation orchestrated efficiently and preserved across Plus-plan interruptions. Avoid recursive subagent fan-out: it consumed the shared window rapidly without advancing M3. The recursively created M3 planning agents all hit the same Plus limit; continue locally until shared agent capacity is restored, then use only bounded leaf reviews. - Full assignment remains **M1–M7 plus C1**, as specified in - [SOL-HANDOFF.md](SOL-HANDOFF.md). M5 is next because it depends on accepted - M3, while the external M4 Claude live gate remains recorded and paused. + [SOL-HANDOFF.md](SOL-HANDOFF.md). M5 and M6 have no remaining independent + work; their external Claude and Jev gates remain recorded while M7 proceeds. - M5 Plan 07-01 has started at `d96e9e4`. The core and Bash compatibility boundaries now include a Claude 2.1.220 headless adapter with structured output, version-scoped model/effort preparation, explicit Read/Glob/Grep or @@ -174,6 +173,24 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne keeps runtime adoption off. The gate is 291 core tests with 2 optional-SDK skips and 220 Bash assertions. M6-D2 remains blocked only on `TYPESAFE_API_KEY`; Laya remains conditional on its declared trigger. +- M7 packaging and the currently available live surfaces are verified through + `ccba125`. The immutable standalone installer works without Claude, performs + offline exact-lock MCP installation with `pip check`, emits clean JSON, + migrates only a recognized legacy launcher, retains old releases and is + idempotent. The active installed release is + `0.1.0-py31214-4c5f034c67bf-mcp-a26bc88afbef`; setup reports all four host + registrations unchanged and doctor is ready. Bundled Codex + 0.155.0-alpha.9.2 passed native initialize plus a complete seven-model + catalog and is preferred over PATH Codex 0.135.0. A terminal-created run was + observed through real Gemini/Antigravity and Codex MCP calls, then cancelled + from terminal. The gate is 300 core tests with 2 optional-SDK skips, 227 Bash + assertions, 8 focused installer tests and 22 installed-SDK tests. See + [M7-STATUS.md](M7-STATUS.md) and the + [portable redacted evidence](evidence/M7-installed-runtime-2026-09-29.json). + The clean-home install/setup/doctor walkthrough passed and exposed the + remaining independent M7 defect: normal `squad review` and `squad fix` + commands are absent. Claude, Grok and the installed two-model delivery remain + external blockers after that task-entry layer is complete. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -200,9 +217,11 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 291 discovered through M6-D1 decision-helper integration; suite OK with 2 optional-SDK skips and ResourceWarning promoted to error | -| Bash 3.2 regression suite | 10 test files, 220 assertions passed | +| Python core discovery | 300 tests passed through M7 Codex binary resolution, with 2 optional-SDK skips and ResourceWarning promoted to error | +| Bash 3.2 regression suite | 11 test files, 227 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | +| M7 installed runtime | Current immutable release has no source/plugin/installed drift; `pip check`, idempotent reinstall, four-host unchanged setup and doctor passed | +| M7 live surface proof | Terminal start/cancel plus real Gemini/Antigravity and ephemeral Codex `squad_status` calls observed the same run/version | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | | M4 cross-surface proof | Real Codex read the terminal-started run through MCP; official SDK clients proved identical ledger, fenced claims, completion and disconnect survival; actual Claude handoff remains blocked on login | | Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 8 | @@ -227,29 +246,28 @@ The successful run retained separate stderr files of 138,030 and the earlier apparent nonresponses. The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md), -[M2-STATUS.md](M2-STATUS.md) and [M3-STATUS.md](M3-STATUS.md). -[backlog.json](backlog.json) marks M1–M3 complete and M4/M5 blocked only on -their external Claude live gates. M6 implementation work remains pending. -Unauthenticated, unsupported or permission-blocked provider paths are not -advertised as verified. +[M2-STATUS.md](M2-STATUS.md), [M3-STATUS.md](M3-STATUS.md), +[M5-STATUS.md](M5-STATUS.md), [M6-STATUS.md](M6-STATUS.md) and +[M7-STATUS.md](M7-STATUS.md). [backlog.json](backlog.json) marks M1–M3 +complete, M4/M5 blocked on Claude, M6 blocked on the missing Jev key, and M7 +in progress. Unauthenticated or unsupported provider paths are not advertised +as verified. ## Exact next work 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Continue M7 with packaging/install/update migration, fresh standalone use, - update idempotency and active-run survival, compatibility fixtures, - quickstart/troubleshooting and supported local-surface receipts. Do not - rebuild the completed M5 offline path or M6 offline implementation. -3. Keep the M4 Claude/Grok/Antigravity probes paused until their normal login - or trust blockers are resolved. Their live gates remain open, but M5 may - proceed independently from accepted M3. -4. Execute the small M6 decision-helper work packages alongside other - independently ready M6 work. Keep experiments off by default and preserve - all existing M4/M5/live acceptance gates. - Exception already authorized: once TypeSafe login/API-key setup is complete, - run the prepared one-request synthetic Jev pilot immediately, save the - private receipt outside Git, and update only redacted aggregate evidence. +2. Implement the missing M7 `squad review` and `squad fix` normal entry layer + over the strict saved-run contracts, using bounded preparation and no hand- + written task JSON. Keep `squad council` scoped to the separate C1 extension. + Do not rebuild the completed M5/M6 offline implementations. +3. Keep Claude and Grok live probes paused until normal login is restored. + Afterward, run the M4 Claude handoff, the M5 installed Claude-to-Codex + delivery and one bounded Grok operation, retaining only redacted evidence. +4. Keep M6 decision guidance off by default. Once `TYPESAFE_API_KEY` is + supplied, run the prepared one-request synthetic Jev pilot immediately with + no retry and the $0.01 ceiling. Install/run Laya only if the predeclared Jev + cost/access/quality trigger fires. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index d8716e1..d8d15f8 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -275,7 +275,7 @@ "id": "M6", "title": "Shared capacity and evidence-based improvement", "depends_on": ["M4", "M5"], - "status": "in_progress", + "status": "blocked", "acceptance_section": "M6 — Make capacity and improvement evidence useful", "decision_helper_work_packages": [ { @@ -387,10 +387,29 @@ "id": "M7", "title": "Installation, migration and all-surface proof", "depends_on": ["M6"], - "status": "pending", + "status": "in_progress", "acceptance_section": "M7 — Package, migrate and prove every requested surface", - "evidence": [], - "blocker": null + "evidence": [ + { + "kind": "packaging_checkpoint", + "revision": "2275b49", + "command_or_action": "299 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 227 Bash assertions, 8 focused installer tests and a fresh offline MCP install", + "outcome": "The Claude-independent immutable installer, stable launcher, idempotent update, payload drift checks, legacy-launcher migration, active-run release retention, one-line JSON and pip dependency validation are verified.", + "artifact": "M7-STATUS.md", + "recorded_at": "2026-09-29T10:34:00Z", + "availability": "tracked_tests" + }, + { + "kind": "installed_runtime_checkpoint", + "revision": "ccba125", + "command_or_action": "300 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 227 Bash assertions, exact Codex 0.155 native handshake/model catalog, four-host unchanged setup and live Terminal/Antigravity/Codex saved-run operations", + "outcome": "The current immutable MCP release is installed with no drift and all four registrations matching. Terminal started and cancelled one saved run; scoped Gemini and ephemeral Codex observed the identical run/version through the installed MCP server. Claude login, Grok authentication and the installed two-model delivery remain open.", + "artifact": "evidence/M7-installed-runtime-2026-09-29.json", + "recorded_at": "2026-09-29T10:52:08Z", + "availability": "portable_redacted" + } + ], + "blocker": "The clean-home install/setup/doctor walkthrough passed, but the required normal squad review/fix entry layer is still independent implementation work. Universal surface completion then requires normal Claude login for the M4/M5 handoff and delivery receipts plus renewed Grok authentication for its live smoke." } ], "extensions": [ diff --git a/docs/plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json b/docs/plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json new file mode 100644 index 0000000..5c03abf --- /dev/null +++ b/docs/plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json @@ -0,0 +1,116 @@ +{ + "schema_version": 1, + "milestone": "M7", + "status": "in_progress", + "implementation_revision": "ccba125e342041ff2ef72fd7e0c9a9b44c356e7d", + "recorded_at": "2026-09-29T10:52:08Z", + "offline_gate": { + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core", + "result": "300 tests passed with 2 optional-SDK skips", + "bash_command": "bash test/run.sh", + "bash_result": "11 test files and 227 assertions passed", + "installer_focus": "8 fresh-install tests passed, including offline synthetic wheels, one-line JSON and pip check", + "installed_sdk_focus": "22 tests passed against the official mcp==2.2.0 environment" + }, + "standalone_install": { + "launcher": "/Users/Dikshant/.local/bin/squad", + "immutable_release": "/Users/Dikshant/.devsquad/releases/0.1.0-py31214-4c5f034c67bf-mcp-a26bc88afbef", + "release_manifest_sha256": "513cb11ff54fae31cb994c2e8931cd7cd86476642c92d92bcc1cbf3d0ed5e5f8", + "source_digest": "4c5f034c67bf1f83a185df0a8910058f38bfdfc7b38e4430923ab1167d08a048", + "mcp_environment_digest": "a26bc88afbef7b5a1ed23014ebece205902a8c43c047fec7afb893da945e66b2", + "python": "3.12.14", + "mcp_sdk": "2.2.0", + "pip_check": "passed", + "second_install": "changed=false with source/plugin/installed drift all false", + "doctor": "ready; four installed local apps and four matching registrations", + "second_setup": "all four hosts unchanged" + }, + "packaging_and_migration": { + "fresh_without_claude": "passed", + "legacy_plugin_contents": "passed", + "legacy_launcher_migration": "known DevSquad release symlink adopted; arbitrary launcher refused", + "active_run_update_survival": "passed with the old immutable release retained", + "duplicate_registration": "second setup changed no host", + "json_contract": "optional MCP install emitted exactly one JSON line", + "dependency_integrity": "installer now runs pip check before publishing an MCP release" + }, + "fresh_home_walkthrough": { + "home": "private temporary directory outside the repository", + "base_install": "documented composite --core-only command succeeded without Claude and squad --version returned 0.1.0", + "mcp_update": "documented offline wheelhouse command selected a second immutable release", + "dry_run": "all four installed hosts returned would_add without mutation", + "setup": "all four hosts returned added and matching inside the temporary home", + "doctor": "ready with four installed/configured hosts and exact mcp==2.2.0", + "setup_dry_run_sha256": "3d282aa80d94e3f7b8acbf32ee94bca21bc37aa6c8ffca4431cfb3a68a7ec2ef", + "setup_sha256": "104aa74642aa8b523b3e14d7a37d66fd57897d1ad4c9ced62b1748df05866ea1", + "doctor_sha256": "15697608ea6d6a1cf9be21017de7d698cfbf1e496947fa20c7da75b2d1cbe936", + "finding": "the required normal squad review/fix/council entry commands are absent; M7 remains in progress while the shared task-entry layer is implemented" + }, + "codex_compatibility": { + "binary": "/Applications/ChatGPT.app/Contents/Resources/codex", + "version": "codex-cli 0.155.0-alpha.9.2", + "native_probe": "initialize and complete model/list succeeded without a model generation; seven models, one default, reasoning-effort metadata present", + "resolver": "verified bundled binary wins over unverified PATH codex-cli 0.135.0", + "status": "supported" + }, + "cross_surface_run": { + "run_id": "f121e96b-502a-4986-a3c3-7ba8ab418bb3", + "terminal_start_state": "awaiting_host", + "terminal_start_version": 14, + "observed_release": "0.1.0-py31214-6a705bf7d65d-mcp-a26bc88afbef", + "terminal_cancel_state": "cancelled", + "terminal_cancel_version": 20, + "handoff_status": "cancelled", + "private_probe_state_sha256": "b44f0efa04b1a3a52f316b530d24bb7b15ffff77785ad98f9a3e46e907eb3c85", + "same_run_id_across_surfaces": true + }, + "antigravity_live_receipt": { + "version": "1.2.13", + "model": "gemini-3.8-flash-low", + "permission": "project-scoped mcp(devsquad/squad_status) only", + "tool": "squad_status", + "tool_result": { + "run_id": "f121e96b-502a-4986-a3c3-7ba8ab418bb3", + "state": "awaiting_host", + "version": 14, + "next_action": "claim_handoff" + }, + "duration_seconds": 42.290051, + "usage": { + "input_tokens": 29629, + "cache_read_tokens": 20388, + "output_tokens": 170, + "thinking_tokens": 0 + }, + "private_result_sha256": "6cab24e6905c5e4e35c2218320b548efc4dfc4835d4e96e21f8720b4d2cb8fec" + }, + "codex_live_receipt": { + "version": "codex-cli 0.155.0-alpha.9.2", + "model": "gpt-6-luna", + "effort": "low", + "session": "ephemeral; user configuration and history ignored; only the DevSquad MCP server configured", + "permission": "read-only sandbox with approval policy never", + "tool": "squad_status", + "tool_result": { + "run_id": "f121e96b-502a-4986-a3c3-7ba8ab418bb3", + "state": "awaiting_host", + "version": 14, + "next_action": "claim_handoff" + }, + "usage": { + "input_tokens": 56834, + "cached_input_tokens": 47360, + "output_tokens": 148, + "reasoning_output_tokens": 0 + }, + "private_events_sha256": "67fb9ba8061f8fe5a8845f53c1330e4dc3ae64e9cd25d62cad6dcf7ed9bb9d9b", + "private_last_message_sha256": "739df4cb90c4b2d1d3604b2456e15ae0527cebd9081d2584929292b95abb3f3a" + }, + "remaining_live_gates": { + "normal_task_entry": "squad review and squad fix are not yet implemented; squad council belongs to the separately gated C1 extension", + "claude_code": "M4 handoff and M5 implementation-to-Codex delivery require normal Claude login; claude auth status reports loggedIn=false", + "grok_build": "registration matches, but provider authentication is expired", + "installed_delivery": "the required genuine installed Claude implementation followed by a different-model Codex review/check/disposition remains blocked with M5", + "jev": "the optional capped M6-D2 measurement remains blocked on TYPESAFE_API_KEY and is not required for normal routing" + } +} From 85aa378cf990687a9889409674bfa530fe4890a7 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 04:15:51 -0700 Subject: [PATCH 123/197] feat: add guided review and fix entry --- docs/generated/core-reference.md | 4 +- plugin/core/schemas/task.schema.json | 2 +- plugin/core/src/devsquad/cli.py | 186 +++++++++- plugin/core/src/devsquad/service.py | 44 ++- plugin/core/src/devsquad/supervisor.py | 10 +- plugin/core/src/devsquad/task_entry.py | 468 +++++++++++++++++++++++++ plugin/core/src/devsquad/validation.py | 23 +- test/core/test_cli.py | 146 ++++++++ test/core/test_service.py | 40 +++ test/core/test_task_entry.py | 278 +++++++++++++++ test/core/test_validation.py | 41 +++ 11 files changed, 1214 insertions(+), 28 deletions(-) create mode 100644 plugin/core/src/devsquad/task_entry.py create mode 100644 test/core/test_task_entry.py diff --git a/docs/generated/core-reference.md b/docs/generated/core-reference.md index 772230a..d966310 100644 --- a/docs/generated/core-reference.md +++ b/docs/generated/core-reference.md @@ -12,6 +12,7 @@ Regenerate with `python3 scripts/generate-core-reference.py`; verify with - `squad classify [-h] [--cwd CWD] [--model MODEL] [--effort EFFORT] [--permission {read_only,workspace_write}] [--timeout TIMEOUT] [--transport {cli_exec,native_protocol}] [--catalog-file CATALOG_FILE] --returncode RETURNCODE --stdout-file STDOUT_FILE --stderr-file STDERR_FILE {codex,antigravity,grok}` - `squad doctor [-h] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` - `squad events [-h] [--after AFTER] [--limit LIMIT] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad fix [-h] [--base BASE] [--target TARGET] [--project-dir PROJECT_DIR] [--write-path WRITE_PATH] [--check CHECK] [--check-timeout CHECK_TIMEOUT] [--review-model REVIEW_MODEL] [--review-effort REVIEW_EFFORT] [--review-mode {standard,adversarial}] [--review-focus REVIEW_FOCUS] [--implementer-model IMPLEMENTER_MODEL] [--implementer-effort IMPLEMENTER_EFFORT] [--idempotency-key IDEMPOTENCY_KEY] [--dry-run] [--wait] [--json] [--runtime-dir RUNTIME_DIR] issue` - `squad handoff [-h] {claim,complete} ...` - `squad handoff claim [-h] --expected-version EXPECTED_VERSION --owner OWNER [--claim-file CLAIM_FILE] [--json] [--runtime-dir RUNTIME_DIR] run` - `squad handoff complete [-h] --claim-file CLAIM_FILE --decision-file DECISION_FILE [--json] [--runtime-dir RUNTIME_DIR] run` @@ -34,6 +35,7 @@ Regenerate with `python3 scripts/generate-core-reference.py`; verify with - `squad report [-h] --project PROJECT [--json] [--runtime-dir RUNTIME_DIR]` - `squad result [-h] [--json] [--runtime-dir RUNTIME_DIR] run` - `squad resume [-h] [--json] [--runtime-dir RUNTIME_DIR] [--recovery-file RECOVERY_FILE] run` +- `squad review [-h] [--base BASE] [--target TARGET] [--project-dir PROJECT_DIR] [--model MODEL] [--effort EFFORT] [--mode {standard,adversarial}] [--focus FOCUS] [--check CHECK] [--check-timeout CHECK_TIMEOUT] [--idempotency-key IDEMPOTENCY_KEY] [--dry-run] [--wait] [--json] [--runtime-dir RUNTIME_DIR]` - `squad setup [-h] [--host {codex,claude-code,antigravity,grok}] [--dry-run] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` - `squad start [-h] --task-file TASK_FILE --idempotency-key IDEMPOTENCY_KEY [--supersedes-run SUPERSEDES_RUN] [--wait] [--json] [--runtime-dir RUNTIME_DIR]` - `squad status [-h] [--json] [--runtime-dir RUNTIME_DIR] run` @@ -68,7 +70,7 @@ runtime never infers them from chat history. | `profile.schema.json` | `https://devsquad.local/schemas/profile-v1.json` | `id`, `harness`, `model_family`, `model_id`, `effort`, `required_tools`, `permission_policy`, `account_pool_id`, `billing_mode`, `quality_status`, `evidence_refs` | `fc51b49cd7dd6a134eedf2c2c941301392aef29a2163ac2d307b4fb78509ec96` | | `profiles.schema.json` | `https://devsquad.local/schemas/profiles-v1.json` | `schema_version`, `profiles`, `bindings` | `c14823c89d1b81ee93b74f5bbc2701ae975825db795b5f0937c28fc057d84cf1` | | `review-result.schema.json` | `https://devsquad.local/schemas/review-result-v1.json` | `schema_version`, `candidate_sha256`, `base_oid`, `target_oid`, `review_mode`, `verdict`, `summary`, `findings` | `96b852543f92dca21dafd6c0bc954f8f56d3dbfb0981d601cc05b9c1c9ba6e6b` | -| `task.schema.json` | `https://devsquad.local/schemas/task-v1.json` | `schema_version`, `project`, `workflow`, `goal`, `task_class`, `acceptance`, `checks`, `scope`, `lead`, `routing`, `budget`, `origin` | `467fcb30221db38c842fcc9eb4c107f3cc90f390de332c1d50576e0db37afc9f` | +| `task.schema.json` | `https://devsquad.local/schemas/task-v1.json` | `schema_version`, `project`, `workflow`, `goal`, `task_class`, `acceptance`, `checks`, `scope`, `lead`, `routing`, `budget`, `origin` | `ec2a700b9dac653b49ffe0bfc6fa24bda92a2a144b0c94e9b66307d93c3bbc19` | ## Task-shape example diff --git a/plugin/core/schemas/task.schema.json b/plugin/core/schemas/task.schema.json index 9b129d5..4413c6c 100644 --- a/plugin/core/schemas/task.schema.json +++ b/plugin/core/schemas/task.schema.json @@ -17,7 +17,7 @@ "scope": {"type":"object","additionalProperties":false,"required":["read_paths","write_paths"],"properties":{"read_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}},"write_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}}}}, "lead": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["host","headless"]}}}, "override": {"type":"object","additionalProperties":false,"required":["profile_id"],"properties":{"profile_id":{"type":"string","minLength":1},"fallback":{"enum":["none","policy"]}}}, - "routing": {"type":"object","additionalProperties":false,"required":["profiles_file","policy_file"],"properties":{"profiles_file":{"type":"string","minLength":1},"policy_file":{"type":"string","minLength":1},"overrides":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"$ref":"#/$defs/override"}}}}, + "routing": {"type":"object","additionalProperties":false,"properties":{"profiles_file":{"type":"string","minLength":1},"policy_file":{"type":"string","minLength":1},"profiles":{"type":"object"},"policy":{"type":"object"},"overrides":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"$ref":"#/$defs/override"}}},"oneOf":[{"required":["profiles_file","policy_file"],"not":{"anyOf":[{"required":["profiles"]},{"required":["policy"]}]}},{"required":["profiles","policy"],"not":{"anyOf":[{"required":["profiles_file"]},{"required":["policy_file"]}]}}]}, "budget": {"type":"object","additionalProperties":false,"required":["wall_seconds","max_worker_invocations","max_revisions","max_fallbacks_per_step"],"properties":{"wall_seconds":{"type":"integer","minimum":1},"max_worker_invocations":{"type":"integer","minimum":1},"max_revisions":{"type":"integer","minimum":0},"max_fallbacks_per_step":{"type":"integer","minimum":0}}}, "origin": {"type":"object","additionalProperties":false,"required":["surface"],"properties":{"surface":{"type":"string","minLength":1},"session_ref":{"type":"string","minLength":1}}}, "review": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["standard","adversarial"]},"focus":{"type":"string","minLength":1}}} diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 9a08f81..3686e92 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -17,6 +17,12 @@ from .integrations import LocalIntegrationManager, load_integrations from .service import Service from .store import ConflictError, SchemaVersionError +from .task_entry import ( + build_managed_task, + discover_codex_identity, + parse_checks, + resolve_repository, +) SOURCE_ROOT = Path(__file__).resolve().parents[2] CORE_ROOT = SOURCE_ROOT if (SOURCE_ROOT / "adapters").is_dir() else Path(sys.prefix) / "share" / "devsquad" @@ -127,13 +133,14 @@ def _service(args: argparse.Namespace) -> Service: return Service(Path(args.runtime_dir)) -def command_start(args: argparse.Namespace) -> tuple[dict, int]: - task = _read_json(args.task_file, "task file") - service = _service(args) - started = service.start(task, args.idempotency_key, args.supersedes_run) - if not args.wait: - return envelope(data=started), 0 +def _wait_for_run( + service: Service, + started: dict[str, Any], + *, + resume_candidate_review: bool = False, +) -> tuple[dict[str, Any], int]: run_id = started["run_id"] + resumed_versions: set[int] = set() try: while True: status = service.status(run_id) @@ -142,6 +149,14 @@ def command_start(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=status), WAIT_EXIT_CODES[state] if state not in WAIT_ACTIVE_STATES: raise RuntimeError(f"service returned unsupported run state: {state!r}") + version = status.get("version") + if (resume_candidate_review + and state == "queued" + and status.get("next_action") == "resume_candidate_review" + and type(version) is int + and version not in resumed_versions): + resumed_versions.add(version) + service.resume(run_id) time.sleep(WAIT_POLL_SECONDS) except KeyboardInterrupt: cancel_command = f"squad cancel {run_id}" @@ -159,6 +174,120 @@ def command_start(args: argparse.Namespace) -> tuple[dict, int]: }), 130 +def command_start(args: argparse.Namespace) -> tuple[dict, int]: + task = _read_json(args.task_file, "task file") + service = _service(args) + started = service.start(task, args.idempotency_key, args.supersedes_run) + if not args.wait: + return envelope(data=started), 0 + return _wait_for_run(service, started) + + +def _normal_entry_result( + summary: dict[str, Any], + idempotency_key: str, + run: dict[str, Any] | None, +) -> dict[str, Any]: + run_id = run.get("run_id") if run is not None else None + state = run.get("state") if run is not None else "not_started" + if run_id is None: + next_action = "rerun this command without --dry-run" + elif state == "succeeded": + next_action = f"squad result {run_id} --json" + else: + next_action = f"squad status {run_id} --json" + return { + "dry_run": run is None, + "run_id": run_id, + "state": state, + "service": run, + "idempotency_key": idempotency_key, + **summary, + "next_action": next_action, + } + + +def _command_normal_entry( + args: argparse.Namespace, + *, + workflow: str, +) -> tuple[dict, int]: + if args.dry_run and args.wait: + raise ContractError("--wait cannot be combined with --dry-run") + repo = resolve_repository(args.project_dir) + codex_identity = discover_codex_identity( + repo, + requested_model=( + args.model if workflow == "branch-review" else args.review_model + ), + requested_effort=( + args.effort if workflow == "branch-review" else args.review_effort + ), + ) + if workflow == "branch-review": + mode = args.mode + focus = args.focus + goal = ( + f"Review exact target {args.target} against base {args.base}" + + (f" with adversarial focus on {focus.strip()}" if focus else "") + + "." + ) + write_paths: tuple[str, ...] = () + claude_model = "sonnet" + claude_effort = "high" + else: + mode = args.review_mode + focus = args.review_focus + goal = args.issue + write_paths = tuple(args.write_path or ()) + claude_model = args.implementer_model + claude_effort = args.implementer_effort + task, summary = build_managed_task( + workflow=workflow, + project_dir=repo, + base_ref=args.base, + target_ref=args.target, + goal=goal, + codex_identity=codex_identity, + write_paths=write_paths, + checks=parse_checks(args.check), + check_timeout=args.check_timeout, + review_mode=mode, + review_focus=focus, + claude_model=claude_model, + claude_effort=claude_effort, + ) + idempotency_key = args.idempotency_key or ( + f"normal-{workflow}-{summary['task_sha256']}" + ) + if args.dry_run: + return envelope(data=_normal_entry_result( + summary, idempotency_key, None, + )), 0 + service = _service(args) + started = service.start(task, idempotency_key, None) + run, code = ( + _wait_for_run( + service, + started, + resume_candidate_review=(workflow == "issue-delivery"), + ) if args.wait + else (envelope(data=started), 0) + ) + service_data = run["data"] + return envelope(data=_normal_entry_result( + summary, idempotency_key, service_data, + )), code + + +def command_review(args: argparse.Namespace) -> tuple[dict, int]: + return _command_normal_entry(args, workflow="branch-review") + + +def command_fix(args: argparse.Namespace) -> tuple[dict, int]: + return _command_normal_entry(args, workflow="issue-delivery") + + def command_capacity_observe(args: argparse.Namespace) -> tuple[dict, int]: observation = _read_json(args.file, "capacity observation file") return envelope(data=_service(args).capacity_observe(observation)), 0 @@ -299,6 +428,51 @@ def parser() -> argparse.ArgumentParser: cmd.add_argument("--stderr-file", required=True) cmd.set_defaults(func=fn) runtime_default = os.environ.get("DEVSQUAD_RUNTIME_DIR", str(Path.home() / ".devsquad" / "runtime")) + review = sub.add_parser( + "review", + help="start an exact-commit Codex branch review without task JSON", + ) + review.add_argument("--base", default="main") + review.add_argument("--target", default="HEAD") + review.add_argument("--project-dir", default=str(Path.cwd())) + review.add_argument("--model") + review.add_argument("--effort") + review.add_argument("--mode", choices=("standard", "adversarial"), default="standard") + review.add_argument("--focus") + review.add_argument("--check", action="append") + review.add_argument("--check-timeout", type=int, default=600) + review.add_argument("--idempotency-key") + review.add_argument("--dry-run", action="store_true") + review.add_argument("--wait", action="store_true") + review.add_argument("--json", action="store_true") + review.add_argument("--runtime-dir", default=runtime_default) + review.set_defaults(func=command_review) + fix = sub.add_parser( + "fix", + help="start bounded Claude implementation and independent Codex review", + ) + fix.add_argument("issue") + fix.add_argument("--base", default="HEAD") + fix.add_argument("--target", default="HEAD") + fix.add_argument("--project-dir", default=str(Path.cwd())) + fix.add_argument("--write-path", action="append") + fix.add_argument("--check", action="append") + fix.add_argument("--check-timeout", type=int, default=600) + fix.add_argument("--review-model") + fix.add_argument("--review-effort") + fix.add_argument( + "--review-mode", choices=("standard", "adversarial"), + default="standard", + ) + fix.add_argument("--review-focus") + fix.add_argument("--implementer-model", default="sonnet") + fix.add_argument("--implementer-effort", default="high") + fix.add_argument("--idempotency-key") + fix.add_argument("--dry-run", action="store_true") + fix.add_argument("--wait", action="store_true") + fix.add_argument("--json", action="store_true") + fix.add_argument("--runtime-dir", default=runtime_default) + fix.set_defaults(func=command_fix) start = sub.add_parser("start"); start.add_argument("--task-file", required=True); start.add_argument("--idempotency-key", required=True); start.add_argument("--supersedes-run"); start.add_argument("--wait", action="store_true"); start.add_argument("--json", action="store_true"); start.add_argument("--runtime-dir", default=runtime_default); start.set_defaults(func=command_start) for name, fn in (("status",command_status),("result",command_result),("cancel",command_cancel),("resume",command_resume)): cmd=sub.add_parser(name); cmd.add_argument("run"); cmd.add_argument("--json",action="store_true"); cmd.add_argument("--runtime-dir",default=runtime_default) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 9a06c13..1f5c20a 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -567,24 +567,40 @@ def _resolve_snapshot( scope_paths = tuple(dict.fromkeys( task["scope"]["read_paths"] + task["scope"]["write_paths"] )) - config_paths = { - label: repo_relative_config(repo, task["routing"][label], label) - for label in ("profiles_file", "policy_file") - } + embedded_routing = "profiles" in task["routing"] + config_paths = ( + {} + if embedded_routing + else { + label: repo_relative_config(repo, task["routing"][label], label) + for label in ("profiles_file", "policy_file") + } + ) if internal_delay is None: assert_clean_inputs(repo, scope_paths, config_paths.values()) configs = {} config_payloads = {} - for label, relative_path in config_paths.items(): - path = repo / relative_path - data = ( - committed_regular_file(repo, target_oid, relative_path) - if internal_delay is None else path.read_bytes() - ) - config_payloads[label] = data - configs[label] = { - "path": str(path), "sha256": hashlib.sha256(data).hexdigest(), - } + if embedded_routing: + for label, routing_label in ( + ("profiles_file", "profiles"), ("policy_file", "policy"), + ): + data = canonical_json(task["routing"][routing_label]).encode() + config_payloads[label] = data + configs[label] = { + "path": f"embedded://routing/{routing_label}", + "sha256": hashlib.sha256(data).hexdigest(), + } + else: + for label, relative_path in config_paths.items(): + path = repo / relative_path + data = ( + committed_regular_file(repo, target_oid, relative_path) + if internal_delay is None else path.read_bytes() + ) + config_payloads[label] = data + configs[label] = { + "path": str(path), "sha256": hashlib.sha256(data).hexdigest(), + } snapshot = { "task": task, "base_oid": base_oid, diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index de4a5c1..ac848b8 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -423,9 +423,13 @@ def _commit_delivery_candidate( scope_paths = tuple(dict.fromkeys( task["scope"]["read_paths"] + task["scope"]["write_paths"] )) - config_paths = tuple( - repo_relative_config(source_repo, task["routing"][label], label) - for label in ("profiles_file", "policy_file") + config_paths = ( + () + if "profiles" in task["routing"] + else tuple( + repo_relative_config(source_repo, task["routing"][label], label) + for label in ("profiles_file", "policy_file") + ) ) review_workspace = prepare_review_workspace( source_repo, diff --git a/plugin/core/src/devsquad/task_entry.py b/plugin/core/src/devsquad/task_entry.py new file mode 100644 index 0000000..94836dc --- /dev/null +++ b/plugin/core/src/devsquad/task_entry.py @@ -0,0 +1,468 @@ +"""Bounded normal-command task preparation over the strict saved-run contract.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path, PurePosixPath +import shlex +import subprocess +import sys +from typing import Any, Iterable + +from .adapters import AdapterManifest, harness_version +from .catalog import normalize_models +from .codex_protocol import ( + JsonLinePeer, + discover_models, + initialize_request, + initialized_notification, + receive_response, +) +from .contracts import ContractError +from .store import canonical_json +from .validation import validate_task + + +SOURCE_ROOT = Path(__file__).resolve().parents[2] +CORE_ROOT = ( + SOURCE_ROOT + if (SOURCE_ROOT / "adapters").is_dir() + else Path(sys.prefix) / "share" / "devsquad" +) +MAX_USER_CHECKS = 12 + + +def _git(repo: Path, *arguments: str) -> str: + try: + completed = subprocess.run( + ["git", "-C", str(repo), *arguments], + text=True, + capture_output=True, + timeout=10, + check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise ContractError("cannot inspect the Git project") from exc + if completed.returncode != 0: + raise ContractError(f"Git project check failed: {' '.join(arguments[:2])}") + return completed.stdout.strip() + + +def resolve_repository(project_dir: str | Path) -> Path: + try: + requested = Path(project_dir).expanduser().resolve(strict=True) + except OSError as exc: + raise ContractError("project directory does not exist") from exc + root = Path(_git(requested, "rev-parse", "--show-toplevel")) + try: + return root.resolve(strict=True) + except OSError as exc: + raise ContractError("Git project root does not exist") from exc + + +def _resolve_commit(repo: Path, reference: str) -> str: + if not isinstance(reference, str) or not reference: + raise ContractError("Git reference must be non-empty") + return _git(repo, "rev-parse", "--verify", f"{reference}^{{commit}}") + + +def _stop(process: subprocess.Popen[str]) -> None: + if process.poll() is None: + process.terminate() + try: + process.wait(timeout=3) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=3) + + +def discover_codex_identity( + repo: Path, + *, + requested_model: str | None = None, + requested_effort: str | None = None, + timeout_seconds: int = 15, +) -> dict[str, str]: + """Discover one currently available exact Codex model without generating.""" + + manifest = AdapterManifest.load(CORE_ROOT / "adapters/codex/adapter.json") + binary = manifest.resolve_binary() + if binary is None: + raise ContractError("Codex is unavailable; run squad doctor") + version = harness_version(binary) + if version not in manifest.verified_versions: + raise ContractError(f"Codex version is not verified: {version or 'unknown'}") + try: + process = subprocess.Popen( + [binary, "app-server", "--listen", "stdio://"], + cwd=repo, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + text=True, + bufsize=1, + start_new_session=True, + ) + except OSError as exc: + raise ContractError("Codex native process could not start") from exc + try: + assert process.stdin is not None and process.stdout is not None + peer = JsonLinePeer(process.stdout, process.stdin) + peer.send(initialize_request(1)) + initialized = receive_response(peer, 1, timeout_seconds=timeout_seconds) + if "error" in initialized: + raise ContractError("Codex native initialization failed") + peer.send(initialized_notification()) + models = normalize_models( + "codex", + version, + discover_models( + peer, first_request_id=10, timeout_seconds=timeout_seconds, + ), + ) + except (EOFError, OSError, TimeoutError) as exc: + raise ContractError("Codex model discovery did not complete") from exc + finally: + _stop(process) + candidates = [ + model for model in models + if model["supported_efforts"] + and (requested_model is None or model["id"] == requested_model) + ] + if not candidates: + qualifier = requested_model or "any model with effort metadata" + raise ContractError(f"Codex model is unavailable: {qualifier}") + selected = next( + (model for model in candidates if model["is_default"]), + candidates[0], + ) + efforts = selected["supported_efforts"] + if requested_effort is not None: + if requested_effort not in efforts: + raise ContractError( + f"Codex effort {requested_effort!r} is unavailable for " + f"{selected['id']}" + ) + effort = requested_effort + else: + effort = next( + (value for value in ("low", "medium") if value in efforts), + efforts[0], + ) + family = selected.get("family") + return { + "harness": "codex", + "harness_version": version, + "model_id": selected["id"], + "model_family": family if isinstance(family, str) and family else "gpt", + "effort": effort, + } + + +def parse_checks(values: Iterable[str] | None) -> tuple[tuple[str, ...], ...]: + parsed = [] + for value in values or (): + if not isinstance(value, str) or not value.strip(): + raise ContractError("check command must be non-empty") + try: + arguments = tuple(shlex.split(value)) + except ValueError as exc: + raise ContractError("check command has invalid quoting") from exc + if not arguments: + raise ContractError("check command must contain an executable") + parsed.append(arguments) + if len(parsed) > MAX_USER_CHECKS: + raise ContractError(f"at most {MAX_USER_CHECKS} check commands are allowed") + return tuple(parsed) + + +def _relative_path(value: str, label: str) -> str: + if not isinstance(value, str) or not value: + raise ContractError(f"{label} must be non-empty") + path = PurePosixPath(value) + if path.is_absolute() or ".." in path.parts: + raise ContractError(f"{label} must be repository-relative") + normalized = path.as_posix().rstrip("/") + return normalized or "." + + +def _managed_routing( + workflow: str, + codex: dict[str, str], + *, + claude_model: str, + claude_effort: str, +) -> dict[str, Any]: + reviewer = { + "id": "managed-codex-reviewer", + "harness": "codex", + "model_family": codex["model_family"], + "model_id": codex["model_id"], + "effort": {"value": codex["effort"], "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "codex-subscription", + "billing_mode": "subscription", + "quality_status": "trial", + "evidence_refs": [ + f"runtime-catalog:{codex['harness_version']}:{codex['model_id']}", + ], + } + profiles = [reviewer] + roles: dict[str, list[dict[str, str]]] = { + "reviewer": [{"kind": "profile", "id": reviewer["id"]}], + } + account_pools: dict[str, dict[str, Any]] = { + "codex-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + }, + } + if workflow == "issue-delivery": + implementer = { + "id": "managed-claude-implementer", + "harness": "claude", + "model_family": "claude-sonnet", + "model_id": claude_model, + "effort": {"value": claude_effort, "transport": "native"}, + "required_tools": ["read", "write"], + "permission_policy": "workspace_write", + "account_pool_id": "claude-subscription", + "billing_mode": "subscription", + "quality_status": "trial", + "evidence_refs": ["verified-claude-cli-2.1.220"], + } + profiles.insert(0, implementer) + roles["implementer"] = [ + {"kind": "profile", "id": implementer["id"]}, + ] + account_pools["claude-subscription"] = { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded", + } + return { + "profiles": { + "schema_version": 1, + "profiles": profiles, + "bindings": {}, + }, + "policy": { + "schema_version": 1, + "id": "managed-normal-entry", + "version": 1, + "roles": roles, + "task_classes": { + "managed-review" if workflow == "branch-review" + else "managed-fix": "trial", + }, + "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, + "account_pools": account_pools, + "experiment_budget": {}, + "decision_helper": {"schema_version": 1, "mode": "off"}, + }, + } + + +def _tracked(repo: Path, relative: str) -> bool: + try: + return bool(_git(repo, "ls-files", "--error-unmatch", "--", relative)) + except ContractError: + return False + + +def _checks( + repo: Path, + *, + workflow: str, + base_oid: str, + target_oid: str, + supplied: tuple[tuple[str, ...], ...], + timeout_seconds: int, +) -> list[dict[str, Any]]: + required = workflow == "issue-delivery" + checks: list[dict[str, Any]] = [{ + "id": "candidate-diff-check", + "argv": ( + ["git", "diff", "--check", base_oid, target_oid, "--"] + if workflow == "branch-review" + else ["git", "diff", "--check", base_oid, "HEAD", "--"] + ), + "cwd": ".", + "timeout_seconds": min(timeout_seconds, 120), + "required_to_pass": required, + }] + detected: tuple[str, ...] | None = None + if _tracked(repo, "test/run.sh"): + detected = ("bash", "test/run.sh") + elif (repo / "tests").is_dir(): + detected = ("python3", "-m", "unittest", "discover", "-s", "tests") + elif (repo / "test").is_dir(): + detected = ("python3", "-m", "unittest", "discover", "-s", "test") + if detected is not None: + checks.append({ + "id": "detected-tests", + "argv": list(detected), + "cwd": ".", + "timeout_seconds": timeout_seconds, + "required_to_pass": required, + }) + for index, arguments in enumerate(supplied, 1): + checks.append({ + "id": f"user-check-{index}", + "argv": list(arguments), + "cwd": ".", + "timeout_seconds": timeout_seconds, + "required_to_pass": required, + }) + if len(checks) > 16: + raise ContractError("normal entry produced too many checks") + return checks + + +def build_managed_task( + *, + workflow: str, + project_dir: str | Path, + base_ref: str, + target_ref: str, + goal: str, + codex_identity: dict[str, str], + write_paths: Iterable[str] = (), + checks: tuple[tuple[str, ...], ...] = (), + check_timeout: int = 600, + review_mode: str = "standard", + review_focus: str | None = None, + claude_model: str = "sonnet", + claude_effort: str = "high", +) -> tuple[dict[str, Any], dict[str, Any]]: + if workflow not in {"branch-review", "issue-delivery"}: + raise ContractError("normal entry workflow is unsupported") + if not isinstance(goal, str) or not goal.strip(): + raise ContractError("goal must be non-empty") + if type(check_timeout) is not int or not 1 <= check_timeout <= 3600: + raise ContractError("check timeout must be between 1 and 3600 seconds") + if review_mode not in {"standard", "adversarial"}: + raise ContractError("review mode must be standard or adversarial") + if review_focus is not None: + if review_mode != "adversarial": + raise ContractError("review focus requires adversarial mode") + if not isinstance(review_focus, str) or not review_focus.strip(): + raise ContractError("review focus must be non-empty") + required_identity = { + "harness", "harness_version", "model_id", "model_family", "effort", + } + if (not isinstance(codex_identity, dict) + or set(codex_identity) != required_identity + or codex_identity.get("harness") != "codex" + or not all( + isinstance(codex_identity[field], str) + and codex_identity[field] + for field in required_identity - {"harness"} + )): + raise ContractError("Codex reviewer identity is invalid") + if workflow == "issue-delivery" and ( + not isinstance(claude_model, str) or not claude_model + or not isinstance(claude_effort, str) or not claude_effort + ): + raise ContractError("Claude model and effort must be non-empty") + repo = resolve_repository(project_dir) + base_oid = _resolve_commit(repo, base_ref) + target_oid = _resolve_commit(repo, target_ref) + normalized_writes = tuple( + dict.fromkeys(_relative_path(path, "write path") for path in write_paths) + ) + if workflow == "branch-review" and normalized_writes: + raise ContractError("review entry cannot declare write paths") + if workflow == "issue-delivery" and not normalized_writes: + normalized_writes = (".",) + routing = _managed_routing( + workflow, + codex_identity, + claude_model=claude_model, + claude_effort=claude_effort, + ) + task_class = "managed-review" if workflow == "branch-review" else "managed-fix" + task: dict[str, Any] = { + "schema_version": 1, + "project": { + "repo_path": str(repo), + "base_ref": base_oid, + "target_ref": target_oid, + }, + "workflow": workflow, + "goal": goal.strip(), + "task_class": task_class, + "acceptance": [ + { + "id": "bounded-goal", + "description": ( + "Review findings are bound to the exact base and target commits." + if workflow == "branch-review" + else f"The candidate addresses this bounded issue: {goal.strip()}" + ), + "evidence_kind": "review", + }, + { + "id": "declared-checks", + "description": "Every declared check result is retained in the receipt.", + "evidence_kind": "check", + }, + ], + "checks": _checks( + repo, + workflow=workflow, + base_oid=base_oid, + target_oid=target_oid, + supplied=checks, + timeout_seconds=check_timeout, + ), + "scope": { + "read_paths": ["."], + "write_paths": list(normalized_writes), + }, + "lead": {"mode": "host"}, + "routing": routing, + "budget": { + "wall_seconds": 900 if workflow == "branch-review" else 1800, + "max_worker_invocations": 1 if workflow == "branch-review" else 7, + "max_revisions": 0 if workflow == "branch-review" else 2, + "max_fallbacks_per_step": 0, + }, + "origin": {"surface": "cli-normal-entry"}, + "review": {"mode": review_mode}, + } + if review_focus is not None: + task["review"]["focus"] = review_focus.strip() + validate_task(task, require_existing_repo=True) + task_sha256 = hashlib.sha256(canonical_json(task).encode()).hexdigest() + roles = { + role: { + "profile_id": profile["id"], + "harness": profile["harness"], + "effort": profile["effort"]["value"], + "permission": profile["permission_policy"], + } + for role, profile in ( + ("implementer", next((p for p in routing["profiles"]["profiles"] if p["harness"] == "claude"), None)), + ("reviewer", next((p for p in routing["profiles"]["profiles"] if p["harness"] == "codex"), None)), + ) + if profile is not None + } + return task, { + "workflow": workflow, + "task_sha256": task_sha256, + "project": str(repo), + "base_oid": base_oid, + "target_oid": target_oid, + "planned_roles": roles, + "selection_reason": ( + "one runtime-discovered subscription profile per required role; " + "different-harness review is mandatory for delivery" + ), + "scope": task["scope"], + "checks": [check["id"] for check in task["checks"]], + } diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py index fc270b7..76928d5 100644 --- a/plugin/core/src/devsquad/validation.py +++ b/plugin/core/src/devsquad/validation.py @@ -70,9 +70,26 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) if value["workflow"] == "branch-review" and scope["write_paths"]: raise ContractError("branch review cannot write") lead = value["lead"]; _exact(lead, {"mode"}, {"mode"}, "lead") if not isinstance(lead["mode"], str) or lead["mode"] not in {"host", "headless"}: raise ContractError("invalid lead mode") - routing = value["routing"]; _exact(routing, {"profiles_file", "policy_file", "overrides"}, {"profiles_file", "policy_file"}, "routing") - for key in ("profiles_file", "policy_file"): - if not isinstance(routing[key], str) or not routing[key]: raise ContractError(f"routing {key} must be a path") + routing = value["routing"] + _exact( + routing, + {"profiles_file", "policy_file", "profiles", "policy", "overrides"}, + set(), + "routing", + ) + file_fields = {"profiles_file", "policy_file"} + embedded_fields = {"profiles", "policy"} + if set(routing) & file_fields == file_fields and not set(routing) & embedded_fields: + for key in file_fields: + if not isinstance(routing[key], str) or not routing[key]: + raise ContractError(f"routing {key} must be a path") + elif set(routing) & embedded_fields == embedded_fields and not set(routing) & file_fields: + validate_profile_registry(routing["profiles"]) + validate_policy(routing["policy"]) + else: + raise ContractError( + "routing requires exactly one complete file or embedded configuration" + ) overrides = routing.get("overrides", {}) if not isinstance(overrides, dict): raise ContractError("routing overrides must be an object") for role, override in overrides.items(): diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 915fd92..e903188 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -74,6 +74,125 @@ def test_start_returns_exact_envelope_and_forwards_inputs(self): service.start.assert_called_once_with({"schema_version": 1}, "key-1", "old-run") service.status.assert_not_called() + def test_review_dry_run_prepares_a_managed_task_without_starting(self): + identity = { + "harness": "codex", "harness_version": "codex fixture", + "model_id": "gpt-fixture", "model_family": "gpt", + "effort": "low", + } + task = {"managed": "review"} + summary = { + "workflow": "branch-review", "task_sha256": "a" * 64, + "project": str(self.root), "base_oid": "b" * 40, + "target_oid": "c" * 40, + "planned_roles": {"reviewer": {"harness": "codex"}}, + "selection_reason": "bounded fixture", "scope": {}, + "checks": ["candidate-diff-check"], + } + with ( + mock.patch.object(cli, "resolve_repository", return_value=self.root) as resolve, + mock.patch.object(cli, "discover_codex_identity", return_value=identity) as discover, + mock.patch.object(cli, "build_managed_task", return_value=(task, summary)) as build, + mock.patch.object(cli, "Service") as service, + ): + code, payload, stderr = self.invoke([ + "review", "--base", "main", "--target", "HEAD", + "--project-dir", str(self.root), "--model", "gpt-fixture", + "--effort", "low", "--check", "python3 -m unittest", + "--dry-run", "--runtime-dir", str(self.runtime), "--json", + ]) + self.assertEqual((code, stderr), (0, "")) + data = payload["data"] + self.assertTrue(data["dry_run"]) + self.assertEqual(data["state"], "not_started") + self.assertIsNone(data["run_id"]) + self.assertEqual( + data["idempotency_key"], + f"normal-branch-review-{'a' * 64}", + ) + self.assertIn("without --dry-run", data["next_action"]) + resolve.assert_called_once_with(str(self.root)) + discover.assert_called_once_with( + self.root, requested_model="gpt-fixture", requested_effort="low", + ) + build.assert_called_once_with( + workflow="branch-review", project_dir=self.root, + base_ref="main", target_ref="HEAD", + goal="Review exact target HEAD against base main.", + codex_identity=identity, write_paths=(), + checks=(("python3", "-m", "unittest"),), check_timeout=600, + review_mode="standard", review_focus=None, + claude_model="sonnet", claude_effort="high", + ) + service.assert_not_called() + + def test_fix_starts_managed_delivery_and_reports_run_and_plan(self): + identity = { + "harness": "codex", "harness_version": "codex fixture", + "model_id": "gpt-review", "model_family": "gpt", + "effort": "high", + } + task = {"managed": "fix"} + summary = { + "workflow": "issue-delivery", "task_sha256": "d" * 64, + "project": str(self.root), "base_oid": "e" * 40, + "target_oid": "e" * 40, + "planned_roles": { + "implementer": {"harness": "claude"}, + "reviewer": {"harness": "codex"}, + }, + "selection_reason": "bounded fixture", + "scope": {"write_paths": ["src"]}, + "checks": ["candidate-diff-check", "user-check-1"], + } + service = mock.Mock() + service.start.return_value = { + "run_id": "run-managed", "state": "queued", "created": True, + } + with ( + mock.patch.object(cli, "resolve_repository", return_value=self.root), + mock.patch.object(cli, "discover_codex_identity", return_value=identity) as discover, + mock.patch.object(cli, "build_managed_task", return_value=(task, summary)) as build, + ): + code, payload, stderr = self.invoke([ + "fix", "Correct the parser edge case.", + "--project-dir", str(self.root), "--write-path", "src", + "--check", "bash test/run.sh", "--review-model", "gpt-review", + "--review-effort", "high", "--review-mode", "adversarial", + "--review-focus", "state transitions", + "--implementer-model", "claude-sonnet-exact", + "--implementer-effort", "high", "--idempotency-key", "fix-1", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + data = payload["data"] + self.assertFalse(data["dry_run"]) + self.assertEqual(data["run_id"], "run-managed") + self.assertEqual(data["state"], "queued") + self.assertEqual(data["planned_roles"], summary["planned_roles"]) + self.assertEqual(data["next_action"], "squad status run-managed --json") + service.start.assert_called_once_with(task, "fix-1", None) + discover.assert_called_once_with( + self.root, requested_model="gpt-review", requested_effort="high", + ) + build.assert_called_once_with( + workflow="issue-delivery", project_dir=self.root, + base_ref="HEAD", target_ref="HEAD", + goal="Correct the parser edge case.", codex_identity=identity, + write_paths=("src",), checks=(("bash", "test/run.sh"),), + check_timeout=600, review_mode="adversarial", + review_focus="state transitions", + claude_model="claude-sonnet-exact", claude_effort="high", + ) + + def test_normal_entry_rejects_waiting_for_a_dry_run(self): + code, payload, _ = self.invoke([ + "review", "--dry-run", "--wait", "--json", + ]) + self.assertEqual(code, 64) + self.assertEqual(payload["error"]["code"], "INPUT_INVALID") + self.assertIn("--wait cannot be combined", payload["error"]["message"]) + def test_status_events_result_cancel_and_resume_operations(self): cases = [ (["status", "run-1"], "status", ("run-1",), {"run_id": "run-1", "state": "failed"}), @@ -452,6 +571,33 @@ def test_wait_observes_active_states_until_terminal(self): self.assertEqual(sleep.call_count, 2) service.cancel.assert_not_called() + def test_managed_fix_wait_resumes_saved_candidate_review_once(self): + service = mock.Mock() + service.status.side_effect = [ + { + "run_id": "run-fix", "state": "queued", "version": 11, + "next_action": "resume_candidate_review", + }, + {"run_id": "run-fix", "state": "running", "version": 13}, + { + "run_id": "run-fix", "state": "awaiting_host", "version": 17, + "next_action": "claim_handoff", + }, + ] + service.resume.return_value = { + "run_id": "run-fix", "disposition": "continued", "launched": True, + } + with mock.patch.object(cli.time, "sleep") as sleep: + response, code = cli._wait_for_run( + service, + {"run_id": "run-fix", "state": "queued"}, + resume_candidate_review=True, + ) + self.assertEqual(code, 2) + self.assertEqual(response["data"]["state"], "awaiting_host") + service.resume.assert_called_once_with("run-fix") + self.assertEqual(sleep.call_count, 2) + def test_wait_keyboard_interrupt_stops_observation_without_cancelling(self): service = mock.Mock() service.start.return_value = {"run_id": "run-42", "state": "running", "created": True} diff --git a/test/core/test_service.py b/test/core/test_service.py index 2398857..4f70380 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -210,6 +210,46 @@ def test_decision_helper_default_off_has_no_runtime_or_call_effect(self): self.assertNotIn("decision_observation", snapshot) self.assertEqual(observations, 0) + def test_embedded_routing_is_hash_frozen_without_repository_config_files(self): + task = json.loads(json.dumps(self.task)) + task["routing"] = { + "profiles": json.loads(self.profiles_json), + "policy": json.loads(self.policy_json), + } + subprocess.run( + ["git", "-C", str(self.repo), "rm", "-q", "profiles.json", "policy.json"], + check=True, + ) + subprocess.run( + ["git", "-C", str(self.repo), "commit", "-qm", "remove routing files"], + check=True, + ) + started = self.service.start( + task, + "embedded-routing", + _internal_review_fixture={ + "verdict": "clean", "summary": "embedded", "findings": [], + }, + ) + self.wait_state(started["run_id"], {"awaiting_host"}) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + try: + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + finally: + store.close() + self.assertEqual( + snapshot["configs"]["profiles_file"]["path"], + "embedded://routing/profiles", + ) + self.assertEqual( + snapshot["configs"]["policy_file"]["path"], + "embedded://routing/policy", + ) + self.assertEqual( + snapshot["routing"]["profile_registry"]["source_sha256"], + snapshot["configs"]["profiles_file"]["sha256"], + ) + def test_shadow_decision_is_cached_and_never_changes_selection(self): self.configure_decision_helper("shadow") fixture = self.decision_fixture() diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py new file mode 100644 index 0000000..0cfa910 --- /dev/null +++ b/test/core/test_task_entry.py @@ -0,0 +1,278 @@ +import io +from pathlib import Path +import subprocess +import sys +import tempfile +import time +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.task_entry import ( + build_managed_task, + discover_codex_identity, + parse_checks, +) + + +class ManagedTaskEntryTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-task-entry-") + self.repo = Path(self.temp.name) / "project" + self.repo.mkdir() + subprocess.run( + ["git", "init", "-b", "main"], cwd=self.repo, + check=True, text=True, capture_output=True, + ) + subprocess.run( + ["git", "config", "user.name", "DevSquad Test"], + cwd=self.repo, check=True, + ) + subprocess.run( + ["git", "config", "user.email", "test@example.invalid"], + cwd=self.repo, check=True, + ) + (self.repo / "src").mkdir() + (self.repo / "src/app.py").write_text("VALUE = 1\n") + (self.repo / "test").mkdir() + (self.repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + subprocess.run( + ["git", "add", "."], cwd=self.repo, check=True, + ) + subprocess.run( + ["git", "commit", "-m", "fixture"], cwd=self.repo, + check=True, text=True, capture_output=True, + ) + self.oid = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=self.repo, + check=True, text=True, capture_output=True, + ).stdout.strip() + self.codex = { + "harness": "codex", + "harness_version": "codex-cli fixture", + "model_id": "gpt-fixture", + "model_family": "gpt", + "effort": "low", + } + + def tearDown(self): + self.temp.cleanup() + + def test_branch_review_freezes_exact_commits_and_embedded_routing(self): + task, summary = build_managed_task( + workflow="branch-review", + project_dir=self.repo / "src", + base_ref="main", + target_ref="HEAD", + goal="Review the exact branch delta.", + codex_identity=self.codex, + checks=parse_checks(["python3 -m unittest"]), + ) + + self.assertEqual(task["project"], { + "repo_path": str(self.repo.resolve()), + "base_ref": self.oid, + "target_ref": self.oid, + }) + self.assertEqual(task["scope"], { + "read_paths": ["."], "write_paths": [], + }) + self.assertNotIn("profiles_file", task["routing"]) + self.assertNotIn("policy_file", task["routing"]) + self.assertEqual( + [profile["harness"] for profile in task["routing"]["profiles"]["profiles"]], + ["codex"], + ) + self.assertTrue(all( + not check["required_to_pass"] for check in task["checks"] + )) + self.assertEqual( + [check["id"] for check in task["checks"]], + ["candidate-diff-check", "detected-tests", "user-check-1"], + ) + self.assertEqual(summary["base_oid"], self.oid) + self.assertEqual( + summary["planned_roles"]["reviewer"]["profile_id"], + "managed-codex-reviewer", + ) + self.assertEqual(len(summary["task_sha256"]), 64) + + def test_issue_delivery_is_bounded_to_one_writer_and_required_checks(self): + task, summary = build_managed_task( + workflow="issue-delivery", + project_dir=self.repo, + base_ref="HEAD", + target_ref="HEAD", + goal="Correct the bounded fixture issue.", + codex_identity=self.codex, + write_paths=("src", "src/"), + checks=parse_checks(["python3 -m unittest discover -s test"]), + review_mode="adversarial", + review_focus="state transitions", + claude_model="claude-sonnet-fixture", + claude_effort="high", + ) + + self.assertEqual(task["scope"]["write_paths"], ["src"]) + self.assertEqual(task["review"], { + "mode": "adversarial", "focus": "state transitions", + }) + self.assertTrue(all( + check["required_to_pass"] for check in task["checks"] + )) + self.assertEqual( + task["checks"][0]["argv"], + ["git", "diff", "--check", self.oid, "HEAD", "--"], + ) + self.assertEqual( + summary["planned_roles"]["implementer"]["profile_id"], + "managed-claude-implementer", + ) + self.assertEqual(summary["planned_roles"]["reviewer"]["harness"], "codex") + self.assertIn("different-harness", summary["selection_reason"]) + + default_task, _ = build_managed_task( + workflow="issue-delivery", + project_dir=self.repo, + base_ref="HEAD", + target_ref="HEAD", + goal="Default scope.", + codex_identity=self.codex, + ) + self.assertEqual(default_task["scope"]["write_paths"], ["."]) + + def test_managed_delivery_runs_offline_through_review_checks_and_handoff(self): + task, _ = build_managed_task( + workflow="issue-delivery", + project_dir=self.repo, + base_ref="HEAD", + target_ref="HEAD", + goal="Change the bounded fixture value.", + codex_identity=self.codex, + write_paths=("src/app.py",), + claude_model="claude-sonnet-fixture", + ) + service = Service(Path(self.temp.name) / "runtime") + started = service.start( + task, + "managed-delivery-offline", + _internal_implementation_fixture={ + "writes": [{"path": "src/app.py", "content": "VALUE = 2\n"}], + "delay_seconds": 0, + }, + _internal_review_fixture={ + "verdict": "clean", + "summary": "The exact managed candidate is clean.", + "findings": [], + }, + ) + deadline = time.monotonic() + 15 + resumed_review = False + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + if status["state"] == "awaiting_host": + break + if (status["state"] == "queued" + and status["next_action"] == "resume_candidate_review" + and not resumed_review): + service.resume(started["run_id"]) + resumed_review = True + if status["state"] in {"failed", "cancelled"}: + self.fail(f"managed delivery terminalized early: {status}") + time.sleep(0.05) + else: + self.fail(f"managed delivery did not reach handoff: {status}") + self.assertEqual(status["next_action"], "claim_handoff") + self.assertEqual((self.repo / "src/app.py").read_text(), "VALUE = 1\n") + + def test_entry_rejects_ambiguous_scope_focus_identity_and_checks(self): + base = { + "workflow": "issue-delivery", + "project_dir": self.repo, + "base_ref": "HEAD", + "target_ref": "HEAD", + "goal": "Bounded issue.", + "codex_identity": self.codex, + } + for writes in (("../outside",), (str(self.repo),)): + with self.subTest(writes=writes), self.assertRaises(ContractError): + build_managed_task(**base, write_paths=writes) + with self.assertRaisesRegex(ContractError, "focus requires adversarial"): + build_managed_task(**base, review_focus="security") + with self.assertRaisesRegex(ContractError, "identity is invalid"): + build_managed_task(**{**base, "codex_identity": {"model_id": "x"}}) + with self.assertRaises(ContractError): + build_managed_task(**base, check_timeout=True) + with self.assertRaisesRegex(ContractError, "cannot declare write paths"): + build_managed_task( + **{**base, "workflow": "branch-review"}, write_paths=("src",), + ) + with self.assertRaisesRegex(ContractError, "invalid quoting"): + parse_checks(['python3 -c "unterminated']) + with self.assertRaisesRegex(ContractError, "at most 12"): + parse_checks(["true"] * 13) + + def test_codex_discovery_selects_requested_exact_model_and_effort(self): + manifest = mock.Mock(verified_versions=("codex-cli fixture",)) + manifest.resolve_binary.return_value = "/fixture/codex" + process = mock.Mock( + stdin=io.StringIO(), stdout=io.StringIO(), + ) + process.poll.return_value = None + models = [ + { + "id": "gpt-default", "family": "gpt", "is_default": True, + "supported_efforts": ["low", "medium"], + }, + { + "id": "gpt-requested", "family": "gpt-6", "is_default": False, + "supported_efforts": ["high", "xhigh"], + }, + ] + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process), + mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), + mock.patch("devsquad.task_entry.receive_response", return_value={"result": {}}), + mock.patch("devsquad.task_entry.discover_models", return_value=[]), + mock.patch("devsquad.task_entry.normalize_models", return_value=models), + ): + identity = discover_codex_identity( + self.repo, + requested_model="gpt-requested", + requested_effort="xhigh", + ) + self.assertEqual(identity, { + "harness": "codex", + "harness_version": "codex-cli fixture", + "model_id": "gpt-requested", + "model_family": "gpt-6", + "effort": "xhigh", + }) + process.terminate.assert_called_once_with() + process.wait.assert_called_once_with(timeout=3) + + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process), + mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), + mock.patch("devsquad.task_entry.receive_response", return_value={"result": {}}), + mock.patch("devsquad.task_entry.discover_models", return_value=[]), + mock.patch("devsquad.task_entry.normalize_models", return_value=models), + ): + with self.assertRaisesRegex(ContractError, "effort 'ultra' is unavailable"): + discover_codex_identity( + self.repo, + requested_model="gpt-requested", + requested_effort="ultra", + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_validation.py b/test/core/test_validation.py index 46e6ca1..5a3091d 100644 --- a/test/core/test_validation.py +++ b/test/core/test_validation.py @@ -65,6 +65,47 @@ def test_task_rejects_adversarial_nested_types(self): value = copy.deepcopy(self.task); mutate(value) self.assert_contract_error(validate_task, value) + def test_task_accepts_exact_embedded_routing_and_rejects_mixed_sources(self): + profiles = { + "schema_version": 1, + "profiles": [{ + "id": "reviewer", "harness": "codex", + "model_family": "gpt", "model_id": "gpt-test", + "effort": {"value": "low", "transport": "native"}, + "required_tools": ["read"], + "permission_policy": "read_only", + "account_pool_id": "codex-subscription", + "billing_mode": "subscription", "quality_status": "trial", + "evidence_refs": ["managed-entry"], + }], + "bindings": {}, + } + policy = { + "schema_version": 1, "id": "managed", "version": 1, + "roles": {"reviewer": [{"kind": "profile", "id": "reviewer"}]}, + "task_classes": {"fixture-bugfix-small": "trial"}, + "require_different_model_for_review": True, + "account_pools": { + "codex-subscription": { + "allowed_billing_modes": ["subscription"], + "max_concurrency": 1, + }, + }, + "experiment_budget": {}, + } + embedded = copy.deepcopy(self.task) + embedded["routing"] = {"profiles": profiles, "policy": policy} + validate_task(embedded) + + mixed = copy.deepcopy(embedded) + mixed["routing"]["profiles_file"] = "profiles.json" + mixed["routing"]["policy_file"] = "policy.json" + self.assert_contract_error(validate_task, mixed) + + partial = copy.deepcopy(embedded) + del partial["routing"]["policy"] + self.assert_contract_error(validate_task, partial) + def test_profile_and_policy_reject_nested_type_confusion(self): profile = {"id":"p","harness":"codex","model_family":"gpt","model_id":"m","effort":{"value":"low","transport":"native"},"required_tools":["read"],"permission_policy":"read_only","account_pool_id":"pool","billing_mode":"subscription","quality_status":"proven","evidence_refs":[]} bad = copy.deepcopy(profile); bad["required_tools"] = [""] From b2c6e34261799cfe9e6b9a181cbe7a55090b76a9 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 04:19:30 -0700 Subject: [PATCH 124/197] docs: close independent M7 work --- docs/RUNTIME-GUIDE.md | 25 ++++++++- docs/plans/engineering-team/M7-STATUS.md | 34 ++++++++----- docs/plans/engineering-team/RESUME.md | 45 ++++++++-------- docs/plans/engineering-team/backlog.json | 13 ++++- .../evidence/M7-normal-entry-2026-09-29.json | 51 +++++++++++++++++++ 5 files changed, 130 insertions(+), 38 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index 50d4b06..608f9ca 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -64,8 +64,27 @@ smoke test. The generated [command and schema reference](generated/core-reference.md) lists every command form, packaged schema digest and a strict task-shape -example. A real task must name an existing Git repository, committed refs and -committed routing files containing locally verified profiles. +example. Normal review and fix entry resolves committed refs, discovers the +installed Codex catalog without a generation, freezes bounded subscription +profiles and embeds the validated routing snapshot. It does not require task, +profile or policy JSON: + +```bash +squad review --base main --wait --json +squad fix "the bounded issue to resolve" \ + --write-path src --check "bash test/run.sh" --wait --json +``` + +Use `--dry-run` first to inspect exact commit IDs, role/profile selections, +scope and checks without creating a run or invoking a model. `squad fix` +defaults to repository-wide write scope when `--write-path` is omitted; narrow +it whenever the issue permits. `--wait` automatically advances the saved +candidate from implementation into independent review, then returns at the +host-lead handoff. Omitting `--wait` returns the run ID immediately. + +The lower-level automation path remains available. A hand-written task must +name an existing Git repository and committed refs. It may use either committed +routing files or an exact embedded registry/policy pair. ```bash squad start --task-file task.json --idempotency-key issue-123 --json @@ -153,6 +172,8 @@ Implemented surfaces and evidence: The current evidence source is [`M7-installed-runtime-2026-09-29.json`](plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json). +The installed normal-entry evidence is +[`M7-normal-entry-2026-09-29.json`](plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json). Claude, Grok and the installed two-model delivery remain open, so universal surface support is not yet claimed. diff --git a/docs/plans/engineering-team/M7-STATUS.md b/docs/plans/engineering-team/M7-STATUS.md index d7e28a5..c49c569 100644 --- a/docs/plans/engineering-team/M7-STATUS.md +++ b/docs/plans/engineering-team/M7-STATUS.md @@ -1,10 +1,10 @@ # M7 implementation status -M7 is **in progress with its independent packaging work complete**. The -standalone runtime is installed and usable from terminal, Codex and -Antigravity. Universal surface completion remains blocked by normal Claude and -Grok authentication, and the installed different-model delivery is the same -external gate retained by M5. +M7 has **completed all independently executable packaging and normal-entry +work**. The standalone runtime is installed and usable from terminal, Codex +and Antigravity. M7 remains blocked on normal Claude and Grok authentication; +the installed different-model delivery is the same external gate retained by +M5. | Requirement | Evidence | Status | |---|---|---| @@ -20,13 +20,13 @@ external gate retained by M5. | Grok Build operation | Registration matching; real operation requires renewed authentication | blocked externally | | Installed two-model delivery | Genuine Claude implementation followed by different-model Codex review/check/disposition | blocked with M5 | | Documentation/CI | Runtime guide, generated command reference and macOS/Python/optional-MCP workflow | clean-home install/setup/doctor passed | -| Normal task entry | `squad review --base ...` and `squad fix "..."` without hand-written task JSON | not implemented; active M7 work | +| Normal task entry | Installed `squad review --base ...` and `squad fix "..."` dry-runs plus an offline end-to-end delivery | verified | ## Current installation The stable launcher is `/Users/Dikshant/.local/bin/squad`. The active immutable release is -`0.1.0-py31214-4c5f034c67bf-mcp-a26bc88afbef`, using Python 3.12.14 and +`0.1.0-py31214-129b9107ed59-mcp-a26bc88afbef`, using Python 3.12.14 and `mcp==2.2.0`. A repeated install reports `changed=false`, `pip check` passes, and `squad setup` reports every host unchanged. @@ -37,12 +37,16 @@ Codex capability drift was revalidated rather than inferred. Bundled ## Test gate -- 300 Python core tests passed with `ResourceWarning` promoted to error; two +- 311 Python core tests passed with `ResourceWarning` promoted to error; two optional-SDK tests skipped in the dependency-free interpreter. - 227/227 Bash assertions passed across 11 files. - Eight focused installer tests passed. - Twenty-two focused tests passed in the installed official MCP environment. - The installed runtime passed `pip check`; setup and doctor report ready. +- The installed `review` and `fix` dry-runs resolved exact Git commits and the + requested verified Codex profile without a model generation. The offline fix + simulation created one isolated writer, froze its candidate, ran independent + review and checks, preserved the source checkout and reached host handoff. The recurring interpreter-finalization `ResourceWarning` was printed as an unraisable cleanup diagnostic during full discovery, but the warning-as-error @@ -57,6 +61,8 @@ action `claim_handoff`. Terminal cancellation then moved the same run and its open handoff to `cancelled`, version 20. Raw provider output remains private; hashes and redacted usage are in [the portable M7 evidence](evidence/M7-installed-runtime-2026-09-29.json). +Normal-entry installation and test details are in +[the normal-entry evidence](evidence/M7-normal-entry-2026-09-29.json). Antigravity headless mode requires an explicit project grant for unattended inspection. The verified minimum is `mcp(devsquad/squad_status)`; no global or @@ -65,12 +71,12 @@ read-only sandbox, no conversation resume and only the DevSquad MCP server. ## Exact remaining work -1. Implement and test the required `squad review` and `squad fix` task-entry - layer over the existing strict saved-run contracts. Keep `squad council` - attached to the separately gated C1 implementation. -2. After normal Claude login, execute the saved M4 handoff and the M5 genuine +1. After normal Claude login, execute the saved M4 handoff and the M5 genuine Claude implementation to different-model Codex delivery. -3. After Grok login renewal, record one supported Grok Build operation against +2. After Grok login renewal, record one supported Grok Build operation against the same installed runtime. -4. Re-run the final gates and mark M7 complete only when every required surface +3. Re-run the final gates and mark M7 complete only when every required surface receipt and installed delivery is present. + +`squad council` remains attached to the separately gated C1 implementation and +does not reopen M7. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 50f059e..5baec6b 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -173,24 +173,30 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne keeps runtime adoption off. The gate is 291 core tests with 2 optional-SDK skips and 220 Bash assertions. M6-D2 remains blocked only on `TYPESAFE_API_KEY`; Laya remains conditional on its declared trigger. -- M7 packaging and the currently available live surfaces are verified through - `ccba125`. The immutable standalone installer works without Claude, performs +- M7 packaging, normal task entry and the currently available live surfaces are + verified through `85aa378`. The immutable standalone installer works without + Claude, performs offline exact-lock MCP installation with `pip check`, emits clean JSON, migrates only a recognized legacy launcher, retains old releases and is idempotent. The active installed release is - `0.1.0-py31214-4c5f034c67bf-mcp-a26bc88afbef`; setup reports all four host + `0.1.0-py31214-129b9107ed59-mcp-a26bc88afbef`; setup reports all four host registrations unchanged and doctor is ready. Bundled Codex 0.155.0-alpha.9.2 passed native initialize plus a complete seven-model catalog and is preferred over PATH Codex 0.135.0. A terminal-created run was observed through real Gemini/Antigravity and Codex MCP calls, then cancelled - from terminal. The gate is 300 core tests with 2 optional-SDK skips, 227 Bash - assertions, 8 focused installer tests and 22 installed-SDK tests. See + from terminal. `85aa378` adds installed `squad review` and `squad fix` + commands that resolve exact commits, discover the Codex catalog without a + generation, embed hash-frozen routing and need no hand-written JSON. The + offline fix gate creates one isolated writer, freezes the candidate, runs + independent review and checks, preserves the source checkout and reaches + host handoff; `--wait` advances the saved candidate-review phase once. The + gate is 311 core tests with 2 optional-SDK skips, 227 Bash assertions, 8 + focused installer tests and 22 installed-SDK tests. See [M7-STATUS.md](M7-STATUS.md) and the - [portable redacted evidence](evidence/M7-installed-runtime-2026-09-29.json). - The clean-home install/setup/doctor walkthrough passed and exposed the - remaining independent M7 defect: normal `squad review` and `squad fix` - commands are absent. Claude, Grok and the installed two-model delivery remain - external blockers after that task-entry layer is complete. + [installed-runtime evidence](evidence/M7-installed-runtime-2026-09-29.json) + plus [normal-entry evidence](evidence/M7-normal-entry-2026-09-29.json). + M7 has no remaining independent work; normal Claude login, renewed Grok + authentication and the installed two-model delivery are external blockers. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -217,10 +223,11 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 300 tests passed through M7 Codex binary resolution, with 2 optional-SDK skips and ResourceWarning promoted to error | +| Python core discovery | 311 tests passed through M7 normal entry, with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 11 test files, 227 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | -| M7 installed runtime | Current immutable release has no source/plugin/installed drift; `pip check`, idempotent reinstall, four-host unchanged setup and doctor passed | +| M7 installed runtime | Current immutable release has no source/plugin/installed drift; `pip check`, idempotent reinstall, four-host unchanged setup, doctor and installed review/fix dry-runs passed | +| M7 normal task entry | Exact-commit review and bounded fix need no hand-written JSON; an offline delivery passed writer, candidate, review, check, source-preservation and handoff gates | | M7 live surface proof | Terminal start/cancel plus real Gemini/Antigravity and ephemeral Codex `squad_status` calls observed the same run/version | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | | M4 cross-surface proof | Real Codex read the terminal-started run through MCP; official SDK clients proved identical ledger, fenced claims, completion and disconnect survival; actual Claude handoff remains blocked on login | @@ -250,24 +257,22 @@ The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md), [M5-STATUS.md](M5-STATUS.md), [M6-STATUS.md](M6-STATUS.md) and [M7-STATUS.md](M7-STATUS.md). [backlog.json](backlog.json) marks M1–M3 complete, M4/M5 blocked on Claude, M6 blocked on the missing Jev key, and M7 -in progress. Unauthenticated or unsupported provider paths are not advertised -as verified. +blocked only on Claude/Grok live receipts. Unauthenticated or unsupported +provider paths are not advertised as verified. ## Exact next work 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Implement the missing M7 `squad review` and `squad fix` normal entry layer - over the strict saved-run contracts, using bounded preparation and no hand- - written task JSON. Keep `squad council` scoped to the separate C1 extension. - Do not rebuild the completed M5/M6 offline implementations. -3. Keep Claude and Grok live probes paused until normal login is restored. +2. Keep Claude and Grok live probes paused until normal login is restored. Afterward, run the M4 Claude handoff, the M5 installed Claude-to-Codex delivery and one bounded Grok operation, retaining only redacted evidence. -4. Keep M6 decision guidance off by default. Once `TYPESAFE_API_KEY` is +3. Keep M6 decision guidance off by default. Once `TYPESAFE_API_KEY` is supplied, run the prepared one-request synthetic Jev pilot immediately with no retry and the $0.01 ceiling. Install/run Laya only if the predeclared Jev cost/access/quality trigger fires. +4. Keep `squad council` scoped to the separately gated C1 extension; it does + not reopen M7 and remains pending after the current M5–M7 goal. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index d8d15f8..17e818e 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -387,7 +387,7 @@ "id": "M7", "title": "Installation, migration and all-surface proof", "depends_on": ["M6"], - "status": "in_progress", + "status": "blocked", "acceptance_section": "M7 — Package, migrate and prove every requested surface", "evidence": [ { @@ -407,9 +407,18 @@ "artifact": "evidence/M7-installed-runtime-2026-09-29.json", "recorded_at": "2026-09-29T10:52:08Z", "availability": "portable_redacted" + }, + { + "kind": "normal_entry_checkpoint", + "revision": "85aa378", + "command_or_action": "311 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 227 Bash assertions, offline end-to-end managed delivery and installed no-generation review/fix dry-runs", + "outcome": "Normal review and bounded fix now require no hand-written task/routing JSON. Exact commits, embedded hash-frozen routing, runtime-discovered Codex selection, one Claude writer, independent Codex review, checks, saved-run continuation and compact next-action output are verified; the updated immutable release passes pip check, idempotent reinstall, four-host setup and doctor.", + "artifact": "evidence/M7-normal-entry-2026-09-29.json", + "recorded_at": "2026-09-29T11:17:22Z", + "availability": "portable_redacted" } ], - "blocker": "The clean-home install/setup/doctor walkthrough passed, but the required normal squad review/fix entry layer is still independent implementation work. Universal surface completion then requires normal Claude login for the M4/M5 handoff and delivery receipts plus renewed Grok authentication for its live smoke." + "blocker": "All independently executable M7 work is complete. Universal surface completion requires normal Claude login for the M4/M5 handoff and installed delivery receipts plus renewed Grok authentication for its live smoke." } ], "extensions": [ diff --git a/docs/plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json b/docs/plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json new file mode 100644 index 0000000..67a92db --- /dev/null +++ b/docs/plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json @@ -0,0 +1,51 @@ +{ + "schema_version": 1, + "milestone": "M7", + "status": "blocked_external", + "implementation_revision": "85aa378cf990687a9889409674bfa530fe4890a7", + "recorded_at": "2026-09-29T11:17:22Z", + "normal_entry": { + "review": "resolves exact base/target commits, uses read-only Codex review and report-only checks, and requires no task/profile/policy JSON", + "fix": "uses one isolated Claude writer, different-harness Codex review, required checks, bounded revisions and host lead disposition", + "routing": "strict embedded profile registry and policy are canonicalized, hash-frozen and retained in the saved run", + "catalog": "Codex model and effort are selected from a complete native model/list without a model generation", + "output": "returns role/profile selection, selection reason, scope, checks, run ID/state and a next command without exposing model IDs in the normal response", + "wait_behavior": "fix --wait resumes the saved candidate-review phase once and returns at the host handoff" + }, + "offline_gate": { + "python_command": "PYTHONWARNINGS=error::ResourceWarning PYTHONDONTWRITEBYTECODE=1 PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -v", + "python_result": "311 tests passed with 2 optional-SDK skips", + "bash_command": "bash test/run.sh", + "bash_result": "11 test files and 227 assertions passed", + "managed_delivery": "one isolated implementation writer, frozen candidate, independent review, required checks, unchanged source checkout and host handoff passed with offline fixtures", + "generated_reference": "current" + }, + "installed_runtime": { + "launcher": "/Users/Dikshant/.local/bin/squad", + "immutable_release": "/Users/Dikshant/.devsquad/releases/0.1.0-py31214-129b9107ed59-mcp-a26bc88afbef", + "release_manifest_sha256": "99debfc9f15c27dc31c13178a91ca4fa05e7aefd75648d3aa519d35ffffa4f63", + "source_digest": "129b9107ed59141d1da8462ce47f16a70f478d49de88a2822a187864816aec74", + "mcp_environment_digest": "a26bc88afbef7b5a1ed23014ebece205902a8c43c047fec7afb893da945e66b2", + "python": "3.12.14", + "mcp_sdk": "2.2.0", + "pip_check": "passed", + "second_install": "changed=false with source/plugin/installed drift all false", + "setup": "all four hosts unchanged and matching", + "doctor": "ready" + }, + "installed_entry_checks": { + "review": { + "command_shape": "squad review --base HEAD^ --target HEAD --model gpt-6-luna --effort low --dry-run --json", + "result": "exact commits, read-only reviewer, detected checks and deterministic idempotency key returned; no run or generation created" + }, + "fix": { + "command_shape": "squad fix ISSUE --write-path PATH --check COMMAND --review-model gpt-6-luna --review-effort low --dry-run --json", + "result": "bounded write scope, Claude implementer, different-harness Codex reviewer, detected/user checks and deterministic idempotency key returned; no run or generation created" + } + }, + "remaining_live_gates": { + "claude_code": "M4 handoff and M5 installed implementation-to-Codex delivery require normal Claude login", + "grok_build": "registration matches, but provider authentication is expired", + "jev": "M6-D2 remains blocked on TYPESAFE_API_KEY; Laya runs only if the declared Jev fallback trigger fires" + } +} From 2d5b450ccf90550fd3e9aef689291a33e2e4d4b1 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 04:28:55 -0700 Subject: [PATCH 125/197] fix: retain completed Codex output after empty delta --- plugin/core/src/devsquad/codex_protocol.py | 2 +- test/core/test_m1.py | 18 ++++++++++++++++++ 2 files changed, 19 insertions(+), 1 deletion(-) diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index 4372744..ad78098 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -171,7 +171,7 @@ def consume(self, message: dict[str, Any]) -> None: raise ContractError("native output delta must be a string") if self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: self.output.append(delta) - if (method == "item/completed" and not self.output + if (method == "item/completed" and not "".join(self.output).strip() and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id): item = params.get("item") diff --git a/test/core/test_m1.py b/test/core/test_m1.py index fed27db..b83f259 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -236,6 +236,24 @@ def test_completed_agent_message_is_output_fallback_without_duplication(self): }) self.assertEqual(state.output, ["fallback"]) + def test_completed_agent_message_replaces_an_empty_stream(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + state.consume({ + "method": "item/agentMessage/delta", + "params": { + "threadId": "th1", "turnId": "t1", "itemId": "item-1", + "delta": "", + }, + }) + state.consume({ + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"type": "agentMessage", "text": "structured result"}, + }, + }) + self.assertEqual("".join(state.output).strip(), "structured result") + def test_unrelated_turn_cannot_complete_ours_and_disconnect_is_visible(self): state = NativeTurnState(thread_id="th1", turn_id="ours") state.consume({"method":"turn/completed", "params":{"threadId":"th1", "turn":{"id":"other", "status":"completed"}}}) From 15537689ec2624c03893cc1b4f24f319c229712e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 04:38:32 -0700 Subject: [PATCH 126/197] fix: recover Codex output from terminal turn --- plugin/core/src/devsquad/codex_protocol.py | 19 +++++++++ test/core/test_m1.py | 49 ++++++++++++++++++++++ 2 files changed, 68 insertions(+) diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index ad78098..582d5b4 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -183,6 +183,25 @@ def consume(self, message: dict[str, Any]) -> None: raise ContractError("native completed agent message must contain text") self.output.append(content) if method == "turn/completed" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: + if not "".join(self.output).strip(): + items = turn.get("items") + if items is not None: + if not isinstance(items, list): + raise ContractError("native terminal turn items must be an array") + completed_output = None + for item in items: + if not isinstance(item, dict): + raise ContractError("native terminal turn item must be an object") + if item.get("type") in {"agentMessage", "agent_message"}: + content = item.get("text", item.get("content")) + if not isinstance(content, str): + raise ContractError( + "native terminal agent message must contain text" + ) + if content.strip(): + completed_output = content + if completed_output is not None: + self.output.append(completed_output) self.terminal = True self.terminal_status = turn.get("status") turn_error = turn.get("error") diff --git a/test/core/test_m1.py b/test/core/test_m1.py index b83f259..73138e1 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -254,6 +254,55 @@ def test_completed_agent_message_replaces_an_empty_stream(self): }) self.assertEqual("".join(state.output).strip(), "structured result") + def test_terminal_turn_items_are_output_fallback_without_duplication(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + state.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": { + "id": "t1", "status": "completed", + "items": [ + {"id": "reasoning", "type": "reasoning", "summary": []}, + {"id": "first", "type": "agentMessage", "text": ""}, + {"id": "final", "type": "agentMessage", "text": "structured result"}, + ], + }, + }, + }) + self.assertEqual(state.output, ["structured result"]) + self.assertTrue(state.terminal) + + streamed = NativeTurnState(thread_id="th1", turn_id="t1") + streamed.consume({ + "method": "item/agentMessage/delta", + "params": {"threadId": "th1", "turnId": "t1", "delta": "streamed"}, + }) + streamed.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": { + "id": "t1", "status": "completed", + "items": [ + {"id": "final", "type": "agentMessage", "text": "duplicate"}, + ], + }, + }, + }) + self.assertEqual(streamed.output, ["streamed"]) + + def test_terminal_turn_rejects_malformed_items(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + with self.assertRaisesRegex(ContractError, "items must be an array"): + state.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": {"id": "t1", "status": "completed", "items": {}}, + }, + }) + def test_unrelated_turn_cannot_complete_ours_and_disconnect_is_visible(self): state = NativeTurnState(thread_id="th1", turn_id="ours") state.consume({"method":"turn/completed", "params":{"threadId":"th1", "turn":{"id":"other", "status":"completed"}}}) From 09e435c098ed9ba3b17ba8b44a5094554a8a2b61 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 04:50:15 -0700 Subject: [PATCH 127/197] fix: report empty Codex protocol output --- plugin/core/src/devsquad/codex_lead_worker.py | 10 ++- plugin/core/src/devsquad/codex_protocol.py | 89 +++++++++++++++++++ .../core/src/devsquad/codex_review_worker.py | 10 ++- test/core/fakes/codex_review_cli.py | 3 +- test/core/test_m1.py | 37 ++++++++ test/core/test_review_runtime.py | 11 ++- 6 files changed, 152 insertions(+), 8 deletions(-) diff --git a/plugin/core/src/devsquad/codex_lead_worker.py b/plugin/core/src/devsquad/codex_lead_worker.py index d4c72db..e676305 100644 --- a/plugin/core/src/devsquad/codex_lead_worker.py +++ b/plugin/core/src/devsquad/codex_lead_worker.py @@ -226,9 +226,13 @@ def record(message: dict[str, Any]) -> None: "native Codex lead did not complete successfully: " f"{state.terminal_status}; {detail}" ) - choice = decode_headless_lead_choice( - "".join(state.output).strip(), handoff["packet"], - ) + lead_payload = "".join(state.output).strip() + if not lead_payload: + raise ContractError( + "native Codex lead completed without output; protocol_summary=" + + canonical_json(state.output_diagnostics()) + ) + choice = decode_headless_lead_choice(lead_payload, handoff["packet"]) usage = _usage(protocol_events, thread_id, turn_id) finally: if process is not None: diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index 582d5b4..fd13821 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -129,6 +129,95 @@ class NativeTurnState: terminal_status: str | None = None error: dict[str, Any] | None = None + def output_diagnostics(self) -> dict[str, Any]: + """Return bounded structural evidence without retaining model text.""" + method_counts: dict[str, int] = {} + matching_deltas = 0 + matching_delta_bytes = 0 + matching_completed_messages = 0 + matching_completed_message_bytes = 0 + matching_terminal_turns = 0 + matching_terminal_agent_messages = 0 + matching_terminal_agent_message_bytes = 0 + terminal_item_types: set[str] = set() + uncorrelated_output_events = 0 + for event in self.events: + method = event.get("method") + if not isinstance(method, str): + method = "" + method_counts[method] = method_counts.get(method, 0) + 1 + params = event.get("params") + if not isinstance(params, dict): + continue + event_turn = params.get("turn") + turn = event_turn if isinstance(event_turn, dict) else {} + message_thread = params.get("threadId") + message_turn = turn.get("id") or params.get("turnId") + correlated = ( + message_thread == self.thread_id and message_turn == self.turn_id + ) + if method in {"item/agentMessage/delta", "turn/output/delta"}: + if not correlated: + uncorrelated_output_events += 1 + continue + delta = params.get("delta") + if isinstance(delta, str): + matching_deltas += 1 + matching_delta_bytes += len(delta.encode("utf-8")) + elif method == "item/completed": + item = params.get("item") + if not isinstance(item, dict) or item.get("type") not in { + "agentMessage", "agent_message", + }: + continue + if not correlated: + uncorrelated_output_events += 1 + continue + content = item.get("text", item.get("content")) + if isinstance(content, str): + matching_completed_messages += 1 + matching_completed_message_bytes += len(content.encode("utf-8")) + elif method == "turn/completed" and correlated: + matching_terminal_turns += 1 + items = turn.get("items") + if not isinstance(items, list): + continue + for item in items: + if not isinstance(item, dict): + terminal_item_types.add("") + continue + item_type = item.get("type") + terminal_item_types.add( + item_type if isinstance(item_type, str) else "" + ) + if item_type not in {"agentMessage", "agent_message"}: + continue + content = item.get("text", item.get("content")) + if isinstance(content, str): + matching_terminal_agent_messages += 1 + matching_terminal_agent_message_bytes += len( + content.encode("utf-8") + ) + return { + "collected_output_bytes": sum( + len(part.encode("utf-8")) for part in self.output + ), + "collected_output_parts": len(self.output), + "event_count": len(self.events), + "matching_completed_message_bytes": matching_completed_message_bytes, + "matching_completed_messages": matching_completed_messages, + "matching_delta_bytes": matching_delta_bytes, + "matching_deltas": matching_deltas, + "matching_terminal_agent_message_bytes": ( + matching_terminal_agent_message_bytes + ), + "matching_terminal_agent_messages": matching_terminal_agent_messages, + "matching_terminal_turns": matching_terminal_turns, + "method_counts": dict(sorted(method_counts.items())), + "terminal_item_types": sorted(terminal_item_types), + "uncorrelated_output_events": uncorrelated_output_events, + } + def consume(self, message: dict[str, Any]) -> None: if not isinstance(message, dict): raise ContractError("native message must be an object") diff --git a/plugin/core/src/devsquad/codex_review_worker.py b/plugin/core/src/devsquad/codex_review_worker.py index 581f735..9b15a12 100644 --- a/plugin/core/src/devsquad/codex_review_worker.py +++ b/plugin/core/src/devsquad/codex_review_worker.py @@ -454,9 +454,13 @@ def record(message: dict[str, Any]) -> None: "native Codex review did not complete successfully: " f"{state.terminal_status}; {detail}" ) - review = decode_review_document( - "".join(state.output).strip(), snapshot["task"], workspace, - ) + review_payload = "".join(state.output).strip() + if not review_payload: + raise ContractError( + "native Codex review completed without output; protocol_summary=" + + canonical_json(state.output_diagnostics()) + ) + review = decode_review_document(review_payload, snapshot["task"], workspace) usage = _usage(protocol_events, thread_id, turn_id) finally: if process is not None: diff --git a/test/core/fakes/codex_review_cli.py b/test/core/fakes/codex_review_cli.py index db5bcf3..6ac5270 100755 --- a/test/core/fakes/codex_review_cli.py +++ b/test/core/fakes/codex_review_cli.py @@ -105,7 +105,8 @@ "findings": [], } output = ( - "{}" if model.endswith("malformed") + "" if model.endswith("empty") + else "{}" if model.endswith("malformed") else json.dumps(review, sort_keys=True, separators=(",", ":")) ) print(json.dumps({ diff --git a/test/core/test_m1.py b/test/core/test_m1.py index 73138e1..4cdf686 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -303,6 +303,43 @@ def test_terminal_turn_rejects_malformed_items(self): }, }) + def test_output_diagnostics_are_structural_and_redacted(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + secret = "do-not-retain-this-model-text" + for message in [ + { + "method": "item/agentMessage/delta", + "params": {"threadId": "other", "turnId": "t1", "delta": secret}, + }, + { + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"type": "agentMessage", "text": secret}, + }, + }, + { + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": { + "id": "t1", "status": "completed", + "items": [{"type": "agentMessage", "text": secret}], + }, + }, + }, + ]: + state.consume(message) + diagnostics = state.output_diagnostics() + self.assertEqual(diagnostics["event_count"], 3) + self.assertEqual(diagnostics["matching_completed_messages"], 1) + self.assertEqual( + diagnostics["matching_completed_message_bytes"], len(secret.encode()), + ) + self.assertEqual(diagnostics["matching_terminal_agent_messages"], 1) + self.assertEqual(diagnostics["uncorrelated_output_events"], 1) + self.assertNotIn(secret, json.dumps(diagnostics, sort_keys=True)) + def test_unrelated_turn_cannot_complete_ours_and_disconnect_is_visible(self): state = NativeTurnState(thread_id="th1", turn_id="ours") state.consume({"method":"turn/completed", "params":{"threadId":"th1", "turn":{"id":"other", "status":"completed"}}}) diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index 4567fb9..34356ea 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -647,7 +647,7 @@ def test_native_codex_faults_never_become_valid_reviews(self): "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", "CODEX_HOME": str(fake_home), } - for mode in ("malformed", "denied", "disconnect", "identity-drift"): + for mode in ("empty", "malformed", "denied", "disconnect", "identity-drift"): with self.subTest(mode=mode): profile["model_id"] = f"gpt-fake-{mode}" (self.repo / "devsquad/profiles.json").write_text( @@ -691,6 +691,15 @@ def test_native_codex_faults_never_become_valid_reviews(self): Path(artifacts["receipt.json"]["path"]).read_bytes(), Path(artifacts["result-receipt.json"]["path"]).read_bytes(), ) + if mode == "empty": + stderr_artifact = next( + artifact for name, artifact in artifacts.items() + if name.endswith(".stderr") + ) + stderr_text = Path(stderr_artifact["path"]).read_text() + self.assertIn("completed without output", stderr_text) + self.assertIn("protocol_summary=", stderr_text) + self.assertIn('"matching_deltas":2', stderr_text) manifest = json.loads( Path(artifacts["artifact-manifest.json"]["path"]).read_text() ) From 9d70888f8e3a3510e41cb83127670d984946649b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 04:58:20 -0700 Subject: [PATCH 128/197] fix: prefer completed Codex agent output --- plugin/core/src/devsquad/codex_lead_worker.py | 12 ++++- plugin/core/src/devsquad/codex_protocol.py | 54 +++++++++++-------- .../core/src/devsquad/codex_review_worker.py | 14 ++++- test/core/fakes/codex_review_cli.py | 34 ++++++++++++ test/core/probes/native_codex_smoke.py | 2 +- test/core/test_m1.py | 39 +++++++++++--- 6 files changed, 122 insertions(+), 33 deletions(-) diff --git a/plugin/core/src/devsquad/codex_lead_worker.py b/plugin/core/src/devsquad/codex_lead_worker.py index e676305..5477b40 100644 --- a/plugin/core/src/devsquad/codex_lead_worker.py +++ b/plugin/core/src/devsquad/codex_lead_worker.py @@ -226,13 +226,21 @@ def record(message: dict[str, Any]) -> None: "native Codex lead did not complete successfully: " f"{state.terminal_status}; {detail}" ) - lead_payload = "".join(state.output).strip() + lead_payload = state.final_output().strip() + if len(lead_payload.encode("utf-8")) > MAX_LEAD_BYTES: + raise ContractError("native Codex lead output exceeds its byte limit") if not lead_payload: raise ContractError( "native Codex lead completed without output; protocol_summary=" + canonical_json(state.output_diagnostics()) ) - choice = decode_headless_lead_choice(lead_payload, handoff["packet"]) + try: + choice = decode_headless_lead_choice(lead_payload, handoff["packet"]) + except ContractError as exc: + raise ContractError( + f"{exc}; protocol_summary=" + + canonical_json(state.output_diagnostics()) + ) from exc usage = _usage(protocol_events, thread_id, turn_id) finally: if process is not None: diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index fd13821..2d17210 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -125,10 +125,17 @@ class NativeTurnState: terminal: bool = False interrupted_acknowledged: bool = False output: list[str] = field(default_factory=list) + completed_output: str | None = None events: list[dict[str, Any]] = field(default_factory=list) terminal_status: str | None = None error: dict[str, Any] | None = None + def final_output(self) -> str: + """Prefer the last authoritative completed agent message over deltas.""" + if self.completed_output is not None: + return self.completed_output + return "".join(self.output) + def output_diagnostics(self) -> dict[str, Any]: """Return bounded structural evidence without retaining model text.""" method_counts: dict[str, int] = {} @@ -203,6 +210,10 @@ def output_diagnostics(self) -> dict[str, Any]: len(part.encode("utf-8")) for part in self.output ), "collected_output_parts": len(self.output), + "completed_output_bytes": ( + len(self.completed_output.encode("utf-8")) + if self.completed_output is not None else None + ), "event_count": len(self.events), "matching_completed_message_bytes": matching_completed_message_bytes, "matching_completed_messages": matching_completed_messages, @@ -260,8 +271,7 @@ def consume(self, message: dict[str, Any]) -> None: raise ContractError("native output delta must be a string") if self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: self.output.append(delta) - if (method == "item/completed" and not "".join(self.output).strip() - and self.thread_id and self.turn_id + if (method == "item/completed" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id): item = params.get("item") if not isinstance(item, dict): @@ -270,27 +280,27 @@ def consume(self, message: dict[str, Any]) -> None: content = item.get("text", item.get("content")) if not isinstance(content, str): raise ContractError("native completed agent message must contain text") - self.output.append(content) + if content.strip(): + self.completed_output = content if method == "turn/completed" and self.thread_id and self.turn_id and message_thread == self.thread_id and message_turn == self.turn_id: - if not "".join(self.output).strip(): - items = turn.get("items") - if items is not None: - if not isinstance(items, list): - raise ContractError("native terminal turn items must be an array") - completed_output = None - for item in items: - if not isinstance(item, dict): - raise ContractError("native terminal turn item must be an object") - if item.get("type") in {"agentMessage", "agent_message"}: - content = item.get("text", item.get("content")) - if not isinstance(content, str): - raise ContractError( - "native terminal agent message must contain text" - ) - if content.strip(): - completed_output = content - if completed_output is not None: - self.output.append(completed_output) + items = turn.get("items") + if items is not None: + if not isinstance(items, list): + raise ContractError("native terminal turn items must be an array") + completed_output = None + for item in items: + if not isinstance(item, dict): + raise ContractError("native terminal turn item must be an object") + if item.get("type") in {"agentMessage", "agent_message"}: + content = item.get("text", item.get("content")) + if not isinstance(content, str): + raise ContractError( + "native terminal agent message must contain text" + ) + if content.strip(): + completed_output = content + if completed_output is not None: + self.completed_output = completed_output self.terminal = True self.terminal_status = turn.get("status") turn_error = turn.get("error") diff --git a/plugin/core/src/devsquad/codex_review_worker.py b/plugin/core/src/devsquad/codex_review_worker.py index 9b15a12..8bece54 100644 --- a/plugin/core/src/devsquad/codex_review_worker.py +++ b/plugin/core/src/devsquad/codex_review_worker.py @@ -454,13 +454,23 @@ def record(message: dict[str, Any]) -> None: "native Codex review did not complete successfully: " f"{state.terminal_status}; {detail}" ) - review_payload = "".join(state.output).strip() + review_payload = state.final_output().strip() + if len(review_payload.encode("utf-8")) > MAX_REVIEW_BYTES: + raise ContractError("native Codex review output exceeds its byte limit") if not review_payload: raise ContractError( "native Codex review completed without output; protocol_summary=" + canonical_json(state.output_diagnostics()) ) - review = decode_review_document(review_payload, snapshot["task"], workspace) + try: + review = decode_review_document( + review_payload, snapshot["task"], workspace, + ) + except ContractError as exc: + raise ContractError( + f"{exc}; protocol_summary=" + + canonical_json(state.output_diagnostics()) + ) from exc usage = _usage(protocol_events, thread_id, turn_id) finally: if process is not None: diff --git a/test/core/fakes/codex_review_cli.py b/test/core/fakes/codex_review_cli.py index 6ac5270..0ba37b9 100755 --- a/test/core/fakes/codex_review_cli.py +++ b/test/core/fakes/codex_review_cli.py @@ -122,6 +122,28 @@ "turn": {"id": turn_id, "status": "inProgress"}, }, }), flush=True) + if model == "gpt-fake-review": + print(json.dumps({ + "method": "item/agentMessage/delta", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "itemId": "fixture-draft", + "delta": '{"draft":true}', + }, + }), flush=True) + print(json.dumps({ + "method": "item/completed", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "item": { + "id": "fixture-draft", + "type": "agentMessage", + "text": '{"draft":true}', + }, + }, + }), flush=True) midpoint = len(output) // 2 for part in (output[:midpoint], output[midpoint:]): print(json.dumps({ @@ -132,6 +154,18 @@ "delta": part, }, }), flush=True) + print(json.dumps({ + "method": "item/completed", + "params": { + "threadId": thread_id, + "turnId": turn_id, + "item": { + "id": "fixture-final", + "type": "agentMessage", + "text": output, + }, + }, + }), flush=True) print(json.dumps({ "method": "thread/tokenUsage/updated", "params": { diff --git a/test/core/probes/native_codex_smoke.py b/test/core/probes/native_codex_smoke.py index bf30809..3b253bf 100644 --- a/test/core/probes/native_codex_smoke.py +++ b/test/core/probes/native_codex_smoke.py @@ -195,7 +195,7 @@ def main() -> int: if remaining <= 0: raise TimeoutError("native turn did not reach correlated terminal state") state.consume(peer.receive(remaining)) - output = "".join(state.output).strip() + output = state.final_output().strip() if state.terminal_status != "completed" or output != "DEVSQUAD_M1_NATIVE_OK": raise RuntimeError(f"native verdict was not successful: status={state.terminal_status!r}, output={output!r}") receipt.update({"status": "passed", "harness_version": version, "model": selected["id"], "effort": effort, "permission": "read_only", "terminal_status": state.terminal_status}) diff --git a/test/core/test_m1.py b/test/core/test_m1.py index 4cdf686..951d14a 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -226,7 +226,7 @@ def test_completed_agent_message_is_output_fallback_without_duplication(self): "item": {"type": "agentMessage", "text": "fallback"}, }, }) - self.assertEqual(state.output, ["fallback"]) + self.assertEqual(state.final_output(), "fallback") state.consume({ "method": "item/completed", "params": { @@ -234,7 +234,7 @@ def test_completed_agent_message_is_output_fallback_without_duplication(self): "item": {"type": "agentMessage", "text": "duplicate"}, }, }) - self.assertEqual(state.output, ["fallback"]) + self.assertEqual(state.final_output(), "duplicate") def test_completed_agent_message_replaces_an_empty_stream(self): state = NativeTurnState(thread_id="th1", turn_id="t1") @@ -252,7 +252,34 @@ def test_completed_agent_message_replaces_an_empty_stream(self): "item": {"type": "agentMessage", "text": "structured result"}, }, }) - self.assertEqual("".join(state.output).strip(), "structured result") + self.assertEqual(state.final_output().strip(), "structured result") + + def test_last_completed_agent_message_wins_over_multiple_delta_streams(self): + state = NativeTurnState(thread_id="th1", turn_id="t1") + for item_id, content in (("first", '{"draft":true}'), ("final", '{"ok":true}')): + state.consume({ + "method": "item/agentMessage/delta", + "params": { + "threadId": "th1", "turnId": "t1", + "itemId": item_id, "delta": content, + }, + }) + state.consume({ + "method": "item/completed", + "params": { + "threadId": "th1", "turnId": "t1", + "item": {"id": item_id, "type": "agentMessage", "text": content}, + }, + }) + state.consume({ + "method": "turn/completed", + "params": { + "threadId": "th1", + "turn": {"id": "t1", "status": "completed", "items": []}, + }, + }) + self.assertEqual("".join(state.output), '{"draft":true}{"ok":true}') + self.assertEqual(state.final_output(), '{"ok":true}') def test_terminal_turn_items_are_output_fallback_without_duplication(self): state = NativeTurnState(thread_id="th1", turn_id="t1") @@ -270,7 +297,7 @@ def test_terminal_turn_items_are_output_fallback_without_duplication(self): }, }, }) - self.assertEqual(state.output, ["structured result"]) + self.assertEqual(state.final_output(), "structured result") self.assertTrue(state.terminal) streamed = NativeTurnState(thread_id="th1", turn_id="t1") @@ -285,12 +312,12 @@ def test_terminal_turn_items_are_output_fallback_without_duplication(self): "turn": { "id": "t1", "status": "completed", "items": [ - {"id": "final", "type": "agentMessage", "text": "duplicate"}, + {"id": "final", "type": "agentMessage", "text": "streamed"}, ], }, }, }) - self.assertEqual(streamed.output, ["streamed"]) + self.assertEqual(streamed.final_output(), "streamed") def test_terminal_turn_rejects_malformed_items(self): state = NativeTurnState(thread_id="th1", turn_id="t1") From b1d52adae6bf11c547acb34f3b78c3d2853211be Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 05:09:25 -0700 Subject: [PATCH 129/197] fix: isolate trusted check home --- plugin/core/src/devsquad/review_worker.py | 62 ++++++++++++----------- test/core/test_review_runtime.py | 39 ++++++++++++++ 2 files changed, 72 insertions(+), 29 deletions(-) diff --git a/plugin/core/src/devsquad/review_worker.py b/plugin/core/src/devsquad/review_worker.py index 18d41b6..297bc1c 100644 --- a/plugin/core/src/devsquad/review_worker.py +++ b/plugin/core/src/devsquad/review_worker.py @@ -8,6 +8,7 @@ from pathlib import Path import subprocess import sys +import tempfile import time from typing import Any @@ -65,38 +66,41 @@ def _run_check( started = time.monotonic() stdout_result, stderr_result = _empty_stream(), _empty_stream() cwd = _safe_check_cwd(check_workspace, check["cwd"]) - try: - process = subprocess.Popen( - check["argv"], - cwd=cwd, - env=os.environ.copy(), - stdin=subprocess.DEVNULL, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - start_new_session=False, - close_fds=True, - ) - except OSError: - status, returncode, error_code = "launch_failed", None, "CLI_ERROR" - else: - assert process.stdout is not None and process.stderr is not None - stdout = BoundedDrain(process.stdout, MAX_PREVIEW_CHARS) - stderr = BoundedDrain(process.stderr, MAX_PREVIEW_CHARS) - stdout.start() - stderr.start() + with tempfile.TemporaryDirectory(prefix="devsquad-check-home-") as check_home: + environment = os.environ.copy() + environment["HOME"] = check_home try: - returncode = process.wait(timeout=check["timeout_seconds"]) - status = "passed" if returncode == 0 else "failed" - error_code = None - except subprocess.TimeoutExpired: - status, returncode, error_code = "timed_out", None, "TIMEOUT" - process.terminate() + process = subprocess.Popen( + check["argv"], + cwd=cwd, + env=environment, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + start_new_session=False, + close_fds=True, + ) + except OSError: + status, returncode, error_code = "launch_failed", None, "CLI_ERROR" + else: + assert process.stdout is not None and process.stderr is not None + stdout = BoundedDrain(process.stdout, MAX_PREVIEW_CHARS) + stderr = BoundedDrain(process.stderr, MAX_PREVIEW_CHARS) + stdout.start() + stderr.start() try: - process.wait(timeout=1) + returncode = process.wait(timeout=check["timeout_seconds"]) + status = "passed" if returncode == 0 else "failed" + error_code = None except subprocess.TimeoutExpired: - process.kill() - process.wait(timeout=2) - stdout_result, stderr_result = _stream_result(stdout), _stream_result(stderr) + status, returncode, error_code = "timed_out", None, "TIMEOUT" + process.terminate() + try: + process.wait(timeout=1) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=2) + stdout_result, stderr_result = _stream_result(stdout), _stream_result(stderr) return { "schema_version": 1, "candidate_sha256": candidate_sha256, diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index 34356ea..2d06f69 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -17,6 +17,7 @@ from devsquad.capacity import derive_pool_capacity from devsquad.contracts import ContractError from devsquad.reports import TERMINAL_REPORT_NAMES, build_handoff_reports +from devsquad.review_worker import _run_check from devsquad.service import Service from devsquad.store import ConflictError, Store, request_hash from devsquad_test_fixtures import branch_review_routing_documents @@ -136,6 +137,44 @@ def start_waiting(self, key): self.assertEqual(waiting["state"], "awaiting_host") return started["run_id"], waiting + def test_each_trusted_check_gets_a_fresh_isolated_home(self): + inherited_home = self.root / "inherited-home" + inherited_home.mkdir() + (inherited_home / "credential-marker").write_text("private\n") + script = ( + "from pathlib import Path; import os,sys; " + "home=Path(os.environ['HOME']); inherited=Path(sys.argv[1]); " + "assert home.is_dir(); assert home != inherited; " + "assert not (home/'credential-marker').exists(); " + "assert not (home/'prior-check-marker').exists(); " + "(home/sys.argv[2]).write_text('created\\n'); print(home)" + ) + results = [] + with patch.dict(os.environ, {"HOME": str(inherited_home)}): + for check_id, marker in ( + ("first-home-check", "prior-check-marker"), + ("second-home-check", "second-check-marker"), + ): + results.append(_run_check( + { + "id": check_id, + "argv": [sys.executable, "-c", script, str(inherited_home), marker], + "cwd": ".", + "timeout_seconds": 10, + "required_to_pass": True, + }, + self.repo.resolve(), + "candidate-sha256", + self.target, + )) + + self.assertEqual([result["status"] for result in results], ["passed", "passed"]) + homes = [result["stdout"]["preview"].strip() for result in results] + self.assertNotEqual(homes[0], homes[1]) + self.assertTrue(all(home != str(inherited_home) for home in homes)) + self.assertTrue(all(not Path(home).exists() for home in homes)) + self.assertEqual((inherited_home / "credential-marker").read_text(), "private\n") + def configure_fixture_headless(self): profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) lead = dict(profiles["profiles"][0]) From f4fa6577e2e891151231c9c7d3180be6e9e23faa Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 05:17:00 -0700 Subject: [PATCH 130/197] docs: record live M7 Codex acceptance --- docs/plans/engineering-team/M6-STATUS.md | 7 +- docs/plans/engineering-team/M7-STATUS.md | 20 ++- docs/plans/engineering-team/RESUME.md | 33 +++-- docs/plans/engineering-team/backlog.json | 9 ++ .../M7-live-codex-review-2026-09-29.json | 128 ++++++++++++++++++ 5 files changed, 178 insertions(+), 19 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/M7-live-codex-review-2026-09-29.json diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md index efb7d93..f3d4ea2 100644 --- a/docs/plans/engineering-team/M6-STATUS.md +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -1,6 +1,6 @@ # M6 implementation status -M6 is **in progress**. Shared capacity, the append-only outcome ledger, +M6 is **blocked with all independent implementation complete**. Shared capacity, the append-only outcome ledger, comparison reports, replay-safe one-variable experiment evaluation, proposal generation, the complete offline profile lifecycle and the default-off typed decision helper are verified. Only the externally blocked Jev measurement and @@ -137,5 +137,6 @@ with 2 optional-SDK skips** and **220/220 Bash assertions**. Keep the one-request Jev M6-D2 gate blocked until `TYPESAFE_API_KEY` is supplied. Do not install or run Laya unless the predeclared Jev access, -cost/usage or measured-quality trigger fires. Continue independent delivery at -M7 packaging, installation, update safety and real-surface usability. +cost/usage or measured-quality trigger fires. M7's independently executable +packaging, installation, update-safety and Codex review work is complete; its +remaining live gates do not substitute for the Jev measurement. diff --git a/docs/plans/engineering-team/M7-STATUS.md b/docs/plans/engineering-team/M7-STATUS.md index c49c569..6c52fe0 100644 --- a/docs/plans/engineering-team/M7-STATUS.md +++ b/docs/plans/engineering-team/M7-STATUS.md @@ -21,12 +21,13 @@ M5. | Installed two-model delivery | Genuine Claude implementation followed by different-model Codex review/check/disposition | blocked with M5 | | Documentation/CI | Runtime guide, generated command reference and macOS/Python/optional-MCP workflow | clean-home install/setup/doctor passed | | Normal task entry | Installed `squad review --base ...` and `squad fix "..."` dry-runs plus an offline end-to-end delivery | verified | +| Live normal Codex review | Exact commit range, verified read-only gpt-5.5/low review, isolated trusted checks and artifact-bound host acceptance | verified live at `b1d52ad` | ## Current installation The stable launcher is `/Users/Dikshant/.local/bin/squad`. The active immutable release is -`0.1.0-py31214-129b9107ed59-mcp-a26bc88afbef`, using Python 3.12.14 and +`0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef`, using Python 3.12.14 and `mcp==2.2.0`. A repeated install reports `changed=false`, `pip check` passes, and `squad setup` reports every host unchanged. @@ -37,7 +38,7 @@ Codex capability drift was revalidated rather than inferred. Bundled ## Test gate -- 311 Python core tests passed with `ResourceWarning` promoted to error; two +- 317 Python core tests passed with `ResourceWarning` promoted to error; two optional-SDK tests skipped in the dependency-free interpreter. - 227/227 Bash assertions passed across 11 files. - Eight focused installer tests passed. @@ -47,6 +48,9 @@ Codex capability drift was revalidated rather than inferred. Bundled requested verified Codex profile without a model generation. The offline fix simulation created one isolated writer, froze its candidate, ran independent review and checks, preserved the source checkout and reached host handoff. +- The installed live `review` path selected verified Codex 0.155/gpt-5.5/low, + returned a clean exact-candidate review, passed both frozen checks and + terminalized `succeeded` after an artifact-bound host acceptance. The recurring interpreter-finalization `ResourceWarning` was printed as an unraisable cleanup diagnostic during full discovery, but the warning-as-error @@ -62,7 +66,17 @@ open handoff to `cancelled`, version 20. Raw provider output remains private; hashes and redacted usage are in [the portable M7 evidence](evidence/M7-installed-runtime-2026-09-29.json). Normal-entry installation and test details are in -[the normal-entry evidence](evidence/M7-normal-entry-2026-09-29.json). +[the normal-entry evidence](evidence/M7-normal-entry-2026-09-29.json). The +subsequent installed live review is recorded in +[the live Codex evidence](evidence/M7-live-codex-review-2026-09-29.json). + +The live review run `c611ad4c-5473-4f05-a870-a136f24464d3` bound +`9d70888..b1d52ad`, observed verified gpt-5.5/low read-only execution, returned +zero findings and passed both `git diff --check` and `bash test/run.sh` inside +the trusted check workspace. Its host disposition accepted the exact four +artifact hashes and terminalized at version 22. This proves the installed +Codex review path; it does not replace M5's still-required genuine Claude +implementation half of the two-harness delivery gate. Antigravity headless mode requires an explicit project grant for unattended inspection. The verified minimum is `mcp(devsquad/squad_status)`; no global or diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 5baec6b..c58c9fa 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -174,12 +174,11 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne skips and 220 Bash assertions. M6-D2 remains blocked only on `TYPESAFE_API_KEY`; Laya remains conditional on its declared trigger. - M7 packaging, normal task entry and the currently available live surfaces are - verified through `85aa378`. The immutable standalone installer works without - Claude, performs - offline exact-lock MCP installation with `pip check`, emits clean JSON, - migrates only a recognized legacy launcher, retains old releases and is - idempotent. The active installed release is - `0.1.0-py31214-129b9107ed59-mcp-a26bc88afbef`; setup reports all four host + verified through `b1d52ad`. The immutable standalone installer works without + Claude, performs offline exact-lock MCP installation with `pip check`, emits + clean JSON, migrates only a recognized legacy launcher, retains old releases + and is idempotent. The active installed release is + `0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef`; setup reports all four host registrations unchanged and doctor is ready. Bundled Codex 0.155.0-alpha.9.2 passed native initialize plus a complete seven-model catalog and is preferred over PATH Codex 0.135.0. A terminal-created run was @@ -189,12 +188,20 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne generation, embed hash-frozen routing and need no hand-written JSON. The offline fix gate creates one isolated writer, freezes the candidate, runs independent review and checks, preserves the source checkout and reaches - host handoff; `--wait` advances the saved candidate-review phase once. The - gate is 311 core tests with 2 optional-SDK skips, 227 Bash assertions, 8 - focused installer tests and 22 installed-SDK tests. See + host handoff; `--wait` advances the saved candidate-review phase once. + `1553768`, `09e435c` and `9d70888` align structured-output recovery with + Codex 0.155's completed agent messages and retain only redacted structural + diagnostics. `b1d52ad` gives every trusted check a fresh temporary HOME so + detached checks work without exposing the user's real home or sharing state. + The exact `9d70888..b1d52ad` candidate then received a verified clean + gpt-5.5/low read-only review; both frozen checks passed and the artifact-bound + host acceptance terminalized succeeded. The gate is 317 core tests with 2 + optional-SDK skips, 227 Bash assertions, 8 focused installer tests and 22 + installed-SDK tests. See [M7-STATUS.md](M7-STATUS.md) and the [installed-runtime evidence](evidence/M7-installed-runtime-2026-09-29.json) - plus [normal-entry evidence](evidence/M7-normal-entry-2026-09-29.json). + plus [normal-entry evidence](evidence/M7-normal-entry-2026-09-29.json) and + [live Codex review evidence](evidence/M7-live-codex-review-2026-09-29.json). M7 has no remaining independent work; normal Claude login, renewed Grok authentication and the installed two-model delivery are external blockers. - The user's Jev/Laya request is evaluated in @@ -223,11 +230,11 @@ Verified at the implementation/evidence checkpoints above: | Check | Result | |---|---| -| Python core discovery | 311 tests passed through M7 normal entry, with 2 optional-SDK skips and ResourceWarning promoted to error | +| Python core discovery | 317 tests passed through the M7 live Codex-review repair, with 2 optional-SDK skips and ResourceWarning promoted to error | | Bash 3.2 regression suite | 11 test files, 227 assertions passed | | Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | -| M7 installed runtime | Current immutable release has no source/plugin/installed drift; `pip check`, idempotent reinstall, four-host unchanged setup, doctor and installed review/fix dry-runs passed | -| M7 normal task entry | Exact-commit review and bounded fix need no hand-written JSON; an offline delivery passed writer, candidate, review, check, source-preservation and handoff gates | +| M7 installed runtime | Current immutable release `0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef` has no source/plugin/installed drift; `pip check`, idempotent reinstall, four-host unchanged setup and doctor passed | +| M7 normal task entry | Exact-commit review and bounded fix need no hand-written JSON; an offline delivery passed writer/candidate/review/check/source-preservation/handoff gates, and a live installed Codex review passed both checks and host acceptance | | M7 live surface proof | Terminal start/cancel plus real Gemini/Antigravity and ephemeral Codex `squad_status` calls observed the same run/version | | M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | | M4 cross-surface proof | Real Codex read the terminal-started run through MCP; official SDK clients proved identical ledger, fenced claims, completion and disconnect survival; actual Claude handoff remains blocked on login | diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 17e818e..e3333fe 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -416,6 +416,15 @@ "artifact": "evidence/M7-normal-entry-2026-09-29.json", "recorded_at": "2026-09-29T11:17:22Z", "availability": "portable_redacted" + }, + { + "kind": "live_normal_review_checkpoint", + "revision": "b1d52ad", + "command_or_action": "317 core tests with ResourceWarning promoted to error (suite OK, 2 optional-SDK skips), 227 Bash assertions, immutable MCP reinstall and one installed exact-commit gpt-5.5/low review with host disposition", + "outcome": "Codex terminal output selection and redacted diagnostics are compatible with 0.155, each trusted check receives a fresh isolated HOME, the exact candidate received a verified clean read-only review, both frozen checks passed and the artifact-bound handoff terminalized succeeded. Claude/Grok live gates remain external.", + "artifact": "evidence/M7-live-codex-review-2026-09-29.json", + "recorded_at": "2026-09-29T12:13:54Z", + "availability": "portable_redacted" } ], "blocker": "All independently executable M7 work is complete. Universal surface completion requires normal Claude login for the M4/M5 handoff and installed delivery receipts plus renewed Grok authentication for its live smoke." diff --git a/docs/plans/engineering-team/evidence/M7-live-codex-review-2026-09-29.json b/docs/plans/engineering-team/evidence/M7-live-codex-review-2026-09-29.json new file mode 100644 index 0000000..48fec8c --- /dev/null +++ b/docs/plans/engineering-team/evidence/M7-live-codex-review-2026-09-29.json @@ -0,0 +1,128 @@ +{ + "schema_version": 1, + "milestone": "M7", + "status": "blocked_external", + "implementation_revision": "b1d52adae6bf11c547acb34f3b78c3d2853211be", + "recorded_at": "2026-09-29T12:13:54Z", + "purpose": "Prove the installed normal-entry Codex review path, including trusted checks in the detached worker environment and an artifact-bound host disposition.", + "repairs": [ + { + "revision": "1553768", + "outcome": "A terminal turn-items fallback recovers completed Codex output when delta delivery is incomplete." + }, + { + "revision": "09e435c", + "outcome": "Empty and malformed output failures retain redacted structural protocol diagnostics without model text." + }, + { + "revision": "9d70888", + "outcome": "The last completed agent message is authoritative, preventing draft and final structured messages from being concatenated." + }, + { + "revision": "b1d52ad", + "outcome": "Every trusted check receives a fresh temporary HOME for compatibility and credential isolation; the home is removed after the check." + } + ], + "offline_gate": { + "python_command": "PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core", + "python_result": "317 tests passed with 2 optional-SDK skips; one interpreter-finalization SQLite ResourceWarning was printed while the process still completed successfully", + "focused_result": "26 review-runtime tests passed, including distinct isolated HOME directories, inherited-marker isolation and cleanup", + "bash_command": "bash test/run.sh", + "bash_result": "11 test files and 227 assertions passed", + "generated_reference": "current" + }, + "installed_runtime": { + "launcher": "/Users/Dikshant/.local/bin/squad", + "immutable_release": "/Users/Dikshant/.devsquad/releases/0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", + "release_manifest_sha256": "a277b2e5f0a7b76610c9da20fe00ac05503c3d7d88d879818a7484bcd01039b9", + "source_digest": "9e5cdea2aa99be819f722f82f6e75dcf4f0ef9cf2936df017ec3c7e66bd562a1", + "mcp_environment_digest": "a26bc88afbef7b5a1ed23014ebece205902a8c43c047fec7afb893da945e66b2", + "python": "3.12.14", + "mcp_sdk": "2.2.0", + "pip_check": "passed", + "second_install": "changed=false with source/plugin/installed drift all false", + "setup": "all four hosts unchanged and matching", + "doctor": "ready" + }, + "live_managed_review": { + "run_id": "c611ad4c-5473-4f05-a870-a136f24464d3", + "exact_range": { + "base_oid": "9d70888f8e3a3510e41cb83127670d984946649b", + "target_oid": "b1d52adae6bf11c547acb34f3b78c3d2853211be", + "candidate_sha256": "a48ea624535452112cb1a71cd6c9d8452578752601fe1b9911485c6b3e18edb9" + }, + "observed_identity": { + "harness": "codex", + "harness_version": "codex-cli 0.155.0-alpha.9.2", + "model_provider": "openai", + "model_id": "gpt-5.5", + "effort": "low", + "permission_policy": "read_only", + "verification": "verified" + }, + "review": { + "verdict": "clean", + "finding_count": 0, + "artifact_id": "ce9edb9d-3596-46cb-b282-eaba707b0552", + "sha256": "5171aacdf618edb0b972be5ea752639548abc298584a478c19f635bfddc25795" + }, + "checks": [ + { + "id": "candidate-diff-check", + "status": "passed", + "returncode": 0, + "duration_ms": 12 + }, + { + "id": "detected-tests", + "command": "bash test/run.sh", + "status": "passed", + "returncode": 0, + "duration_ms": 17162, + "result": "11 test files and 227 assertions passed under the isolated trusted-check HOME" + } + ], + "checks_artifact": { + "artifact_id": "f7a49e89-5c5a-48ea-8de7-8ac6545cd71c", + "sha256": "8155b124099979e60f3a02cd3e56eb4d31c099157e8388a71272d7045ba69ae8" + }, + "evaluation_artifact": { + "artifact_id": "87c5669c-7f32-4422-8108-9dc65ca1aacd", + "sha256": "47433aa1c55053d36fe2fca0a49e93fbb7e7780c2cbc8dd33b7ba41729a51602", + "accept_allowed": true, + "required_checks_passed": true + }, + "attempt_artifact": { + "artifact_id": "82277739-0922-4cde-8cfe-6585aa7f96b6", + "sha256": "888961f482d368ee125b40f7841a8af2673d72b07099682a5be2d45838a50b6d" + }, + "native_usage": { + "input_tokens": 119440, + "output_tokens": 2786, + "total_tokens": 122226, + "source": "native_reported" + }, + "handoff": { + "packet_sha256": "0810249a84da5d0537f40e84ee532fd812bfeda231c870c3becb764d41d18d9b", + "disposition": "accept", + "submission_id": "m7-review-b1d52ad-accept", + "submission_hash": "cc3dde0d760d4e1ab140ea1722ab3a9baf3c22c3e822d2fa15858c5ba4a78185" + }, + "terminal": { + "state": "succeeded", + "version": 22, + "receipt_sha256": "0103db19a528cff916ebef80c6ec94b682ed4801fc22db020cdbc21796040f70" + } + }, + "pre_fix_live_evidence": { + "run_id": "228db373-ff4f-4a1d-a692-7dff255036b5", + "receipt_sha256": "2431590ad0bbe4b4bff840f6f673f73671164236cb3933513838e0f65465bce6", + "outcome": "The review itself was clean, but the detected Bash suite failed because the detached trusted-check environment omitted HOME. The host rejected that evidence before b1d52ad." + }, + "remaining_live_gates": { + "claude_code": "Normal Claude login is still required for the M4 handoff and genuine M5 Claude implementation followed by different-model Codex review/check/disposition.", + "grok_build": "Registration matches, but a supported live operation requires renewed authentication.", + "jev": "M6-D2 remains blocked on TYPESAFE_API_KEY; Laya runs only if the declared Jev fallback trigger fires." + }, + "redaction": "Raw provider messages, private diagnostics and credentials remain outside tracked evidence." +} From 399d93d21a52946c8e43856fdb8297ccf7a0a43f Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 07:57:21 -0700 Subject: [PATCH 131/197] WIP checkpoint: Sol review remediation plan and corrected milestone status (2026-09-29 07:57) --- docs/plans/engineering-team/M3-STATUS.md | 9 +- docs/plans/engineering-team/M5-STATUS.md | 27 +- docs/plans/engineering-team/M6-STATUS.md | 35 +- docs/plans/engineering-team/M7-STATUS.md | 25 +- docs/plans/engineering-team/RESUME.md | 76 +++-- docs/plans/engineering-team/SOL-HANDOFF.md | 6 + .../engineering-team/SOL-REVIEW-FOLLOWUP.md | 302 ++++++++++++++++++ docs/plans/engineering-team/START-HERE.md | 17 +- docs/plans/engineering-team/backlog.json | 38 ++- 9 files changed, 457 insertions(+), 78 deletions(-) create mode 100644 docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md diff --git a/docs/plans/engineering-team/M3-STATUS.md b/docs/plans/engineering-team/M3-STATUS.md index 12227d4..8a7d0f6 100644 --- a/docs/plans/engineering-team/M3-STATUS.md +++ b/docs/plans/engineering-team/M3-STATUS.md @@ -1,6 +1,11 @@ # M3 implementation status -M3 is **complete** at implementation checkpoint `1737667`. The final gate +M3 is **reopened for candidate-integrity repair R1** after the September 29 +review at `f4fa657`. A check that changes tracked source can produce passing +evidence for the original candidate. See F1 and its required public-service +regression in [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). + +M3 was accepted at implementation checkpoint `1737667`. That historical gate passes 188 core tests with `ResourceWarning` promoted to an error and all 10 Bash regression files/202 assertions. The bounded live Codex review evidence remains the successful subscription-backed run recorded at `9478796`; no @@ -11,7 +16,7 @@ additional provider turn was used for the closeout. | Deterministic selection | Versioned profile aliases, exact pins, explicit `none`/`policy` fallback, permission/billing filters and typed capacity produce stable frozen routing snapshots | verified offline | | Frozen branch input | Base/target refs resolve to exact OIDs; committed config hashes, candidate hash and detached review/check workspaces remain stable while the submitted checkout, index and HEAD are preserved | verified offline | | Reviewer evidence | Strict ordinary/adversarial prompts, review/check/evaluation schemas, candidate binding, read-only identity checks and malformed/denied/disconnected output faults prevent unsupported success | verified offline | -| Check and lead gates | Report-only failure stays visible without blocking acceptance; required failure blocks acceptance; host and headless leads support accept/reject/revise with bounded revisions | verified offline | +| Check and lead gates | Report-only/required failures and bounded dispositions have historical tests; source-changing checks expose a candidate-integrity gap | reopened: R1 | | Durable host handoff | Waiting JSON/Markdown packets, claim leases, renewal/takeover, stale completion fencing, replay and crash-resume continuation share the saved run ledger | verified offline | | Terminal reporting | Success, rejection, preflight failure, worker failure, cancellation, waiting cancellation, timeout and budget exhaustion publish receipt JSON/Markdown, events, manifest and result receipt | verified offline | | Headless leadership | Offline and native headless leads run as separate fenced attempts, verify their own frozen identity/evidence and terminalize without host intervention | verified offline | diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index 4fbdf40..6610cbf 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -1,22 +1,23 @@ # M5 implementation status -M5 is **blocked on its external live gate**. All independently executable -offline implementation and fault-injection work is complete; the milestone -remains incomplete until a normally authenticated Claude implementation and a -different-model Codex review pass the live two-harness gate. +M5 is **in progress with independent repairs available**. The September 29 +review at `f4fa657` reopened candidate integrity (F1), observed Claude identity +(F2) and complete normal-command check coverage (G4). Execute R1/R2/R6 in +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The genuine Claude-to-Codex +live gate additionally remains blocked on normal Claude login. | Requirement | Planned evidence | Status | |---|---|---| -| Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | verified offline at `d96e9e4` | +| Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | argv conformance verified at `d96e9e4`; observed identity reopened under R2 | | Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | verified offline at `0e88d73` | | Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `0e88d73` | -| Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | workflow and identity gate verified offline at `b7d90cc`; live proof pending | -| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | verified offline through two distinct candidates at `4c76887` | +| Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | reopened under R2; prior fixtures do not detect fabricated observed identity; live proof pending | +| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | reopened under R1/R6 for check mutation and normal-command check coverage | | Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | correction/budgets verified at `4c76887`; same-permission delivery fallback verified at `b7d90cc` | -| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | verified offline at `a1199c6` and `b7d90cc` | +| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | historical failure gates pass; R1/R2 must reject invalid candidate/identity evidence | | Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | success, repair, fallback, failure and cancellation history verified offline at `b7d90cc` | | Crash recovery | Killing a live implementation supervisor cannot create a duplicate writer on resume | prelaunch and live revised-writer recovery verified offline at `f199cd2` | -| Live acceptance | One bounded issue completes across at least two authenticated subscription harnesses with different verified models | blocked on normal Claude CLI login | +| Live acceptance | One bounded issue completes across at least two authenticated subscription harnesses with different verified models | pending repairs and normal Claude CLI login | ## Boundary @@ -136,7 +137,7 @@ passes **241 core tests with 2 optional-SDK skips** and ResourceWarning promoted to error. The compatibility gate remains **220/220 Bash assertions**; the last code change after that run added only the delivery-specific Python regression. -No independent M5 implementation work remains. The unresolved acceptance gate -is intentionally not replaced with fixture evidence: normal Claude CLI login -is required to run a genuine bounded implementation followed by different- -model Codex review/check/disposition and save the redacted two-harness receipt. +This checkpoint's independent-completion assessment was superseded by the +September 29 review. Repair F1/F2/G4 before the genuine implementation and +different-model review/check/disposition gate. Normal Claude login remains +required for that redacted two-harness receipt. diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md index f3d4ea2..0a0a6a4 100644 --- a/docs/plans/engineering-team/M6-STATUS.md +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -1,24 +1,26 @@ # M6 implementation status -M6 is **blocked with all independent implementation complete**. Shared capacity, the append-only outcome ledger, -comparison reports, replay-safe one-variable experiment evaluation, proposal -generation, the complete offline profile lifecycle and the default-off typed -decision helper are verified. Only the externally blocked Jev measurement and -its conditional Laya follow-up remain open. +M6 is **in progress with independent repairs and integration work available**. +The component tests below remain useful historical evidence, but the +September 29 review found reused held-out evidence (F3), normal routing that +bypasses lifecycle bindings (F4), and missing public outcome/experiment, +catalog and quota connections (G1/G2). Execute R3–R5 in +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The Jev key blocks only its +separate measurement; it does not block this engineering work. | Requirement | Planned evidence | Status | |---|---|---| | Strict capacity evidence | Typed pool/window/scope/source/confidence/TTL validation; stale, estimated and incomplete measurements remain unknown | verified at `21b1331` | | Shared transactional reservations | Two projects share one pool; one-slot races produce one owner; schema-8 active attempts survive migration; ambiguous ownership retains the reservation | verified at `26ce5cf` | | Capacity-aware routing | All applicable windows and profile sublimits affect deterministic selection; a later exhausted observation blocks reservation; paid API remains policy-gated | verified at `d170c01` | -| Public observation/status surface | `squad capacity observe --file FILE`; replay-safe persistence; status shows frozen and current detailed evidence | verified at `d170c01` | -| Final and late outcomes | Preserve attempt contribution, lead repair, final success and escaped-defect corrections without crediting failed attempts | verified at `d621df2` | -| Comparison reports | Sample sizes, missingness and separated automatic/pinned/experimental evidence | verified at `45ebc9e` | -| Frozen experiment evaluation | One-variable paired evaluation/held-out cases, failure evidence, no-change or promotion-proposal verdict and rollback target; evaluation never changes active policy | verified at `edfb3f3` | +| Public observation/status surface | `squad capacity observe --file FILE`; replay-safe persistence; status shows frozen and current detailed evidence | manual path verified at `d170c01`; native ingestion pending R4 | +| Final and late outcomes | Preserve attempt contribution, lead repair, final success and escaped-defect corrections without crediting failed attempts | manual components verified at `d621df2`; terminal projection pending R5 | +| Comparison reports | Sample sizes, missingness and separated automatic/pinned/experimental evidence | components verified at `45ebc9e`; public runtime evidence pending R5 | +| Frozen experiment evaluation | One-variable paired evaluation/held-out cases, failure evidence, no-change or promotion-proposal verdict and rollback target; evaluation never changes active policy | reopened under R3 for reused/non-comparable evidence and R5 for real assignments | | Draft proposals | `learn propose` emits content-addressed JSON/Markdown with hashes, sample sizes, missingness, failures and rollback; no evidence yields no-change | verified at `98c6685` | -| Held-out rerun and rollback | Post-change held-out evidence and exercised rollback through lifecycle bindings | verified at `4b27e0c` | -| Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | verified through `ca4ee73` | -| Catalog drift and unavailable incumbent | Complete catalog drift scopes revalidation; added models stay unqualified; removed incumbents roll back only to a prior proven/qualified binding or block | verified at `398ae6a` / `ca4ee73` | +| Held-out rerun and rollback | Post-change held-out evidence and exercised rollback through lifecycle bindings | historical components at `4b27e0c`; evidence/runtime chain reopened under R3/R5 | +| Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | components through `ca4ee73`; qualification integrity and normal alias routing reopened under R3/R4 | +| Catalog drift and unavailable incumbent | Complete catalog drift scopes revalidation; added models stay unqualified; removed incumbents roll back only to a prior proven/qualified binding or block | component tests at `398ae6a` / `ca4ee73`; production discovery connection pending R4 | | Decision helper M6-D1 | Default-off typed contract, fake adapter, cache/accounting and authority/integrity tests | verified at `87fa9cf` | | Jev M6-D2 | One capped synthetic request with exact model/usage/latency/cost receipt | blocked on `TYPESAFE_API_KEY` | | Laya M6-D3 | Triggered pinned local comparison and measured keep-off/adopt decision | pending; run only if the declared Jev trigger fires | @@ -135,8 +137,7 @@ with 2 optional-SDK skips** and **220/220 Bash assertions**. ## Exact next slice -Keep the one-request Jev M6-D2 gate blocked until `TYPESAFE_API_KEY` is -supplied. Do not install or run Laya unless the predeclared Jev access, -cost/usage or measured-quality trigger fires. M7's independently executable -packaging, installation, update-safety and Codex review work is complete; its -remaining live gates do not substitute for the Jev measurement. +Begin R3's failing repeated-outcome regression, then wire R4/R5 through public +saved runs. Keep the one-request Jev M6-D2 gate blocked until +`TYPESAFE_API_KEY` is supplied. Do not install or run Laya unless its declared +trigger fires. A synthetic pilot does not prove production routing quality. diff --git a/docs/plans/engineering-team/M7-STATUS.md b/docs/plans/engineering-team/M7-STATUS.md index 6c52fe0..dc0a670 100644 --- a/docs/plans/engineering-team/M7-STATUS.md +++ b/docs/plans/engineering-team/M7-STATUS.md @@ -1,10 +1,11 @@ # M7 implementation status -M7 has **completed all independently executable packaging and normal-entry -work**. The standalone runtime is installed and usable from terminal, Codex -and Antigravity. M7 remains blocked on normal Claude and Grok authentication; -the installed different-model delivery is the same external gate retained by -M5. +M7 is **in progress with independent normal-entry and readiness work**. +Installation and the recorded Codex/Gemini operations work, but the September +29 review found routing, terminal handoff, authentication-readiness and check- +discovery gaps (F4/G3/G4). Execute R4/R6/R8 in +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). Claude and Grok authentication +remain additional blockers for the corresponding live proofs. | Requirement | Evidence | Status | |---|---|---| @@ -18,9 +19,9 @@ M5. | Antigravity operation | Gemini 3.8 Flash Low called installed `squad_status` with one project-scoped grant | verified live | | Claude Code operation | Registration matching; real operation requires normal login | blocked externally | | Grok Build operation | Registration matching; real operation requires renewed authentication | blocked externally | -| Installed two-model delivery | Genuine Claude implementation followed by different-model Codex review/check/disposition | blocked with M5 | -| Documentation/CI | Runtime guide, generated command reference and macOS/Python/optional-MCP workflow | clean-home install/setup/doctor passed | -| Normal task entry | Installed `squad review --base ...` and `squad fix "..."` dry-runs plus an offline end-to-end delivery | verified | +| Installed two-model delivery | Genuine Claude implementation followed by different-model Codex review/check/disposition | pending M5 repairs and Claude authentication | +| Documentation/CI | Runtime guide, generated command reference and macOS/Python/optional-MCP workflow | historical installation checks pass; readiness/quickstart updates pending R6 | +| Normal task entry | Installed `squad review --base ...` and `squad fix "..."` dry-runs plus an offline end-to-end delivery | reopened under R4/R6 for approved routing, terminal completion and check coverage | | Live normal Codex review | Exact commit range, verified read-only gpt-5.5/low review, isolated trusted checks and artifact-bound host acceptance | verified live at `b1d52ad` | ## Current installation @@ -85,11 +86,13 @@ read-only sandbox, no conversation resume and only the DevSquad MCP server. ## Exact remaining work -1. After normal Claude login, execute the saved M4 handoff and the M5 genuine +1. Complete R4/R6 normal routing, terminal disposition, optional run resolution, + authentication readiness and committed-target check discovery offline. +2. After normal Claude login, execute the saved M4 handoff and the M5 genuine Claude implementation to different-model Codex delivery. -2. After Grok login renewal, record one supported Grok Build operation against +3. After Grok login renewal, record one supported Grok Build operation against the same installed runtime. -3. Re-run the final gates and mark M7 complete only when every required surface +4. Run the affected final gates and mark M7 complete only when every required surface receipt and installed delivery is present. `squad council` remains attached to the separately gated C1 implementation and diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index c58c9fa..0fc8c9e 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -4,6 +4,35 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ## Current position — September 29, 2026 +### Review correction and next action + +The review of `f4fa6577e2e891151231c9c7d3180be6e9e23faa` supersedes the earlier +claim that only credentials remain. M1/M2 remain accepted; M3 is reopened for +candidate integrity (F1); M4 retains its real-Claude-host gate; M5–M7 have +independent repairs and integration work; C1 remains pending full-delivery +scope. No runtime repairs were made during the review or this planning update. + +Execute [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), starting at **R1**: +reproduce source-mutating trusted checks through the public service and prevent +acceptance of invalid candidate evidence. Then repair Claude identity (R2), +independent experiment evidence (R3), routing/catalog/quota (R4), learning +runtime connections (R5), and normal terminal/readiness/check discovery (R6). +R7 covers C1 and R8 covers installed/live closure. Authentication and the Jev +key block their specific live subgates, not the independent engineering work. + +Fresh review verification ran 317 Python tests successfully (2 optional-SDK +skips), 227 Bash assertions, generated-reference validation and an installed +payload comparison. An unclosed SQLite `ResourceWarning` still appeared and +is assigned to R5. Existing live receipts remain evidence of their exact runs, +not proof that the newly identified failure cases are safe. See the follow-up +plan for reproduction details and evidence limits, and +[backlog.json](backlog.json) for current work-package dependencies. + +### Preserved implementation checkpoints + +The following records describe earlier implementation and verification; +the review correction above governs current completion and next work. + - Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. - Build branch: `codex/engineering-team`. `main` remains the published runtime baseline. Inspect current refs before acting; later build checkpoints may be @@ -22,7 +51,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne bounded target. `ddb6f51` fixes both: lease authorization now samples time after acquiring the SQLite write transaction, and public cancel resumes an interrupted `recovery_cleanup`. Both have deterministic regressions. -- M3 is accepted at `1737667`. The branch-review path now includes frozen +- M3 was accepted at `1737667` and is now reopened for F1. The branch-review path includes frozen routing/input, native and offline reviewers, trusted checks, fenced host and headless lead disposition, complete waiting/terminal reports, cumulative budgets, transactional pool capacity and bounded frozen fallbacks. The final @@ -75,8 +104,8 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne planning agents all hit the same Plus limit; continue locally until shared agent capacity is restored, then use only bounded leaf reviews. - Full assignment remains **M1–M7 plus C1**, as specified in - [SOL-HANDOFF.md](SOL-HANDOFF.md). M5 and M6 have no remaining independent - work; their external Claude and Jev gates remain recorded while M7 proceeds. + [SOL-HANDOFF.md](SOL-HANDOFF.md). The current review work packages R1–R8 + replace the earlier credentials-only assessment of M5–M7. - M5 Plan 07-01 has started at `d96e9e4`. The core and Bash compatibility boundaries now include a Claude 2.1.220 headless adapter with structured output, version-scoped model/effort preparation, explicit Read/Glob/Grep or @@ -120,16 +149,16 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne tests (2 optional-SDK skips) and 220 Bash assertions. Delivery fallback, failure/cancellation history, live Claude execution and the live two-harness proof remain open. -- M5 has completed all independently executable offline work at `f199cd2`. +- M5 was assessed as offline-complete at `f199cd2`; F1/F2/G4 now reopen that assessment. `b7d90cc` freezes the real Claude implementation bridge, records observed identity/session/usage, enforces the same-permission rate-limit fallback and preserves delivery failure/cancellation history. `f199cd2` kills a live revised-implementation supervisor and proves retained ownership, no duplicate writer, successful reap and unchanged source checkout/remotes. The exact core gate is 241 tests with 2 optional-SDK skips and ResourceWarning promoted to - error; the compatibility gate is 220 Bash assertions. M5 is now blocked only - on normal Claude login for the genuine Claude implementation → different- - model Codex review/check/disposition receipt. + error; the compatibility gate is 220 Bash assertions. Normal Claude login + remains required for the genuine Claude implementation → different-model + Codex review/check/disposition receipt, after the independent repairs. - M6 shared-capacity work is verified through `d170c01`. Schema 9 persists strict scoped observations and reservations, backfills active schema-8 attempts and keeps ambiguous ownership in flight. Two separate projects @@ -160,8 +189,8 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne additions unqualified, and `ca4ee73` rolls a removed incumbent only to the newest prior proven/qualified profile under the same template or blocks without mutation. The exact gate is 275 core tests with 2 optional-SDK skips - and 220 Bash assertions. Only the default-off decision helper and its - external Jev/Laya measurement path remain open in M6. + and 220 Bash assertions. These component checkpoints do not close the + evidence-integrity and missing runtime connections now assigned to R3–R5. - M6-D1 is verified at `87fa9cf`. The optional decision helper defaults to off; shadow records without changing execution, and advisory can only reorder the deterministic router's already-eligible profiles under reviewed gate @@ -202,8 +231,9 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne [installed-runtime evidence](evidence/M7-installed-runtime-2026-09-29.json) plus [normal-entry evidence](evidence/M7-normal-entry-2026-09-29.json) and [live Codex review evidence](evidence/M7-live-codex-review-2026-09-29.json). - M7 has no remaining independent work; normal Claude login, renewed Grok - authentication and the installed two-model delivery are external blockers. + R4/R6 now identify remaining independent work; normal Claude login, renewed + Grok authentication and the installed two-model delivery are additional + external gates. - The user's Jev/Laya request is evaluated in [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped @@ -218,7 +248,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne Bash assertions. The live call is blocked because the TypeSafe console is at login and no `TYPESAFE_API_KEY` exists. Classifier suggestions never become permission/acceptance authority. - After the bounded probe, M5 remains the implementation priority. + Do not wait for this probe to execute the review repairs starting at R1. ## Completed and preserved @@ -262,24 +292,30 @@ the earlier apparent nonresponses. The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md), [M2-STATUS.md](M2-STATUS.md), [M3-STATUS.md](M3-STATUS.md), [M5-STATUS.md](M5-STATUS.md), [M6-STATUS.md](M6-STATUS.md) and -[M7-STATUS.md](M7-STATUS.md). [backlog.json](backlog.json) marks M1–M3 -complete, M4/M5 blocked on Claude, M6 blocked on the missing Jev key, and M7 -blocked only on Claude/Grok live receipts. Unauthenticated or unsupported -provider paths are not advertised as verified. +[M7-STATUS.md](M7-STATUS.md). [backlog.json](backlog.json) marks M1/M2 complete, +M3/M5/M6/M7 in progress, M4 blocked on its real Claude handoff, and C1 pending. +The review correction and [Sol follow-up plan](SOL-REVIEW-FOLLOWUP.md) govern +where historical verification is incomplete. Unauthenticated or unsupported +provider paths must not be advertised as verified. ## Exact next work 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Keep Claude and Grok live probes paused until normal login is restored. +2. Execute R1's failing public-service candidate-integrity regression, repair + the shared review/delivery gate and checkpoint its evidence. Continue R2/R3, + then R4–R6 in dependency order. Preserve later implementation and historical + receipts; do not rewrite the architecture or reset completed work. +3. Keep Claude and Grok live probes paused until normal login is restored. Afterward, run the M4 Claude handoff, the M5 installed Claude-to-Codex delivery and one bounded Grok operation, retaining only redacted evidence. -3. Keep M6 decision guidance off by default. Once `TYPESAFE_API_KEY` is +4. Keep M6 decision guidance off by default. Once `TYPESAFE_API_KEY` is supplied, run the prepared one-request synthetic Jev pilot immediately with no retry and the $0.01 ceiling. Install/run Laya only if the predeclared Jev cost/access/quality trigger fires. -4. Keep `squad council` scoped to the separately gated C1 extension; it does - not reopen M7 and remains pending after the current M5–M7 goal. +5. Complete the separately gated C1 extension under R7 and audit installed/live + closure under R8. C1 is required in the full assignment even though it does + not reopen M7. Record each blocked subgate without pausing unrelated work. The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. diff --git a/docs/plans/engineering-team/SOL-HANDOFF.md b/docs/plans/engineering-team/SOL-HANDOFF.md index ee2b8a6..de1750e 100644 --- a/docs/plans/engineering-team/SOL-HANDOFF.md +++ b/docs/plans/engineering-team/SOL-HANDOFF.md @@ -1,5 +1,11 @@ # Sol execution handoff — build, test and make DevSquad usable +**Current continuation:** Start with +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The September 29 review at +`f4fa657` reopened shared M3 acceptance integrity and independent M5–M7 work. +That plan supplies the current findings, ordered repairs and acceptance gates; +the full scope and constraints below remain in force. + This is the full execution prompt for Sol. It is an implementation assignment; the underlying runtime is still pending at handoff creation. Copy this document into Sol, or ask Sol to read this file and execute it in full. ## Objective and persistence diff --git a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md new file mode 100644 index 0000000..2626157 --- /dev/null +++ b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md @@ -0,0 +1,302 @@ +# Sol execution plan and review feedback — September 29, 2026 + +## Assignment and starting point + +Continue the existing DevSquad build on `codex/engineering-team`. Repair the +review findings below, finish the missing runtime connections, and prove the +requested product through the public service and installed normal commands. +This is a continuation of [SOL-HANDOFF.md](SOL-HANDOFF.md), not a redesign. +The full delivery remains M1–M7 plus C1; the immediate engineering priority is +the shared acceptance repair followed by M5–M7. C1 is optional to invoke but +required to implement under the original assignment. + +Review baseline: `f4fa6577e2e891151231c9c7d3180be6e9e23faa`. Compare Git state +before acting and retain later work. The review made no source changes. +Installed release at review: +`0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef`, with no detected payload drift. + +Read `AGENTS.md`, `CONTRIBUTING.md`, [RESUME.md](RESUME.md), this document and +[backlog.json](backlog.json) first. Load the relevant contract/specification +for each work package. Do not reload the historical chat or restart M1/M2. +This review supersedes the earlier claim that only credentials remain. + +## Feedback on what has been built + +Preserve the substantial working implementation: the durable runner and +ledger, process ownership and recovery, isolated worktrees, fenced host +handoffs, Codex protocol repairs, immutable installer and MCP integrations. +The successful installed Codex review receipt is real and remains valid for +that exact run. Its SHA256 was rechecked: +`0103db19a528cff916ebef80c6ec94b682ed4801fc22db020cdbc21796040f70`. + +The main problem is the strength of completion claims. Several components +were marked complete because their unit fixtures passed, even though the +normal command path bypasses them or cannot produce their input evidence. +Passing counts do not demonstrate that the complete workflow enforces its +contract. Each new completion claim needs an actual runtime caller, a public +path regression and the corresponding installed/live proof where required. + +During this review, the complete Python suite ran 317 tests successfully +(two optional-SDK skips), and all 227 Bash assertions passed. The generated +reference and installed-source comparison passed. A recurring unclosed +SQLite `ResourceWarning` appeared despite process exit zero; it remains an +open cleanup issue. Initial sandbox-restricted runs were interrupted because +process inspection was denied; the successful reruns had the process access +needed by the cancellation tests. Do not count the interrupted runs as passes. + +### Findings and evidence limits + +Source locations refer to the baseline above and will move as repairs land. + +| ID | Priority / affected scope | Observed behavior and evidence | Repair package | +|---|---|---|---| +| F1 | P1 / M3, M5 | `review_worker.py:169` never revalidates candidate source after checks. A real-Git direct worker/evaluator reproduction froze `VALUE='wrong'`, changed it to `fixed` in a required check, then returned `passed` and `accept_allowed:true` while the frozen candidate stayed wrong. The full service reproduction was environment-blocked, so terminal service acceptance is not claimed. Violates CONTRACTS §5. | R1 | +| F2 | P1 / M5 | `claude_delivery_worker.py:256` copies requested model/effort into observed identity and labels it verified. A mocked result reporting `different-model` still yielded requested-model/low/verified; absent model identity also passes. `workflows.py:572` checks only that identity is a dictionary. | R2 | +| F3 | P1 / M6 | `learning.py:396–414` checks unique case IDs but permits reuse of outcome IDs. A confirmed evaluator reproduction reused two outcomes across two evaluation cases and one held-out case and returned `promotion_proposal`. Qualification reads those inflated counts. | R3 | +| F4 | P1 / M6, M7 | `task_entry.py:136` selects the provider default or first model; `_managed_routing` uses singleton concrete profiles and empty bindings. `store.py:2588` skips lifecycle overlay without aliases. Catalog default changes can change normal routing without qualification. | R4 | +| G1 | Required integration / M6 | The only production read of `experiment_assignment` is in `store.py:1522`; there is no writer. Public runs cannot supply the experimental outcomes required by the evaluator. `record_outcome` is reached through manual `outcome_add`, not ordinary terminal completion. | R5 | +| G2 | Required integration / M6 | Last-good catalog persistence has no production caller. Normal discovery has no scoped persistent refresh/cache path. Native Codex quota observations are not ingested; observations require manual input. | R4 | +| G3 | Required usability / M7 | Normal tasks always use `lead.mode=host`; terminal completion requires manual claim/decision JSON. `status`/`result` require an ID. Doctor checks binaries/registrations, not provider authentication. All CLI responses are JSON even without `--json`. | R6 | +| G4 | Verification coverage / M5, M7 | Default check detection on DevSquad selects only `bash test/run.sh`; the Python core suite is omitted, including when the requested fix changes Python core code. Detection also inspects the current checkout rather than the selected target tree. | R6 | +| G5 | Remaining full-delivery scope / C1 | `squad council` is absent. Its implementation and acceptance are included in the full assignment. | R7 | + +The optional decision helper currently executes only a fake adapter; +non-fixture requests record `unavailable`. This is consistent with the narrow +M6-D1 contract, not proof of a production Jev integration. The one-request +Jev pilot is separate from runtime adoption. Do not enable a classifier or +claim routing improvement from a synthetic smoke result. + +## Work order and ownership + +R1, R2 and R3 come first. R4 precedes R5; R6 integrates the repaired public +paths. R7 follows the relevant M3–M6 repairs. R8 proves the installed product; +its authentication and classifier subgates may remain externally blocked. +Continue all independent work when a live subgate is blocked. + +Use one coordinator and small commits. A bounded independent review may run +alongside non-overlapping work, but do not recursively spawn agents or run +multiple full suites concurrently. `service.py`, `store.py`, `workflows.py` +and the ledger documents require a single integration owner. Prefer focused +offline regressions before a full gate and one justified live probe. + +### R1 — Preserve candidate identity through trusted checks + +**Files:** `review_worker.py`, `workspaces.py`, `workflows.py`, and review/ +delivery runtime tests. Shared M3/M5 fix; no provider dependency. + +1. Turn F1 into a regression through the public service for both review and + delivery. Retain the exact pre-check candidate, check output and mutation. +2. Revalidate HEAD/index/tracked content and candidate inputs across each check + boundary. A tracked edit, deletion, mode change or checkout cannot provide + passing evidence for the original candidate or contaminate the next check. + Distinguish ordinary permitted build outputs from candidate source changes. +3. Make the integrity failure visible in evaluation, handoff and terminal + reports; both host and headless disposition must refuse acceptance. + +**Acceptance:** source-changing check cannot accept; a second check cannot +validate an unnoticed changed tree; ordinary non-source build artifacts and +clean checks work; the original checkout is unchanged; recovery/replay cannot +restore invalidated evidence. Preserve per-check temporary HOME behavior. + +### R2 — Observe and validate Claude execution identity + +**Files:** `claude_delivery_worker.py`, `workflows.py`, adapter fixtures and +delivery tests. Implement conformance offline before spending a live turn. + +1. Verify the supported CLI's native model/session/usage fields against its + installed schema/help and current official documentation when necessary. + Record requested settings separately from actual reported identity. +2. Parse effective model identity, including alias resolution and multiple + model entries where the CLI reports them. Preserve unavailable effort or + backing revision as unknown. Never manufacture an observed setting from + the requested argv. +3. Validate imported identity and enforce the existing independent-review + contract. Missing, contradictory or insufficient identity evidence must + not qualify as verified independence. Keep failed evidence in the receipt. + +**Acceptance:** matching exact identity, alias-to-effective model, unexpected +model, absent identity, multiple models, unavailable effort and tampered +evidence all have explicit tested results. The genuine two-harness live gate +remains open until the corrected adapter obtains a real receipt. + +### R3 — Make experiment and held-out evidence independent + +**Files:** `learning.py`, experiment/qualification paths in `store.py`, and +learning/lifecycle tests. No key or provider required. + +1. Reject duplicate outcomes across cases/arms/splits. Persist enough frozen + case, run and split provenance to prevent the same underlying observation + being relabelled to inflate sample size or contaminate held-out evidence. +2. Validate that each arm belongs to its declared project, experiment, + concrete profile fingerprint and paired input contract. Review arms require + the same candidate; implementation arms require the same baseline, task + and checks. An experimental label alone is insufficient to establish a + controlled one-variable comparison. +3. Enforce these checks before evaluation persistence and qualification. + Audit any saved qualifying evaluations affected by the bug; preserve them + as historical evidence and prevent unsafe future use rather than deleting + or silently rewriting receipts. Late escaped-defect corrections must + trigger an explicit evaluation/qualification revision or review policy, + not leave stale success silently eligible for future promotion. + +**Acceptance:** the two-outcome/three-case reproduction is rejected; repeated +run evidence, crossed split membership, wrong-profile arms and mismatched +tasks fail; valid disjoint pairs qualify; failures/missingness remain visible; +replay cannot bypass the repaired gate. Promotion and rollback use the same +validated evidence path. + +### R4 — Connect normal routing, catalog lifecycle and quota + +**Files:** `task_entry.py`, `catalog.py`, `capacity.py`, `router.py`, +`service.py`, `store.py`, adapter metadata and relevant tests. + +1. Resolve ordinary roles through approved stable aliases and existing + policy. Preserve explicit concrete pins and qualified fallbacks. Bootstrap + without evidence must remain an explicitly labelled bounded trial and + cannot become a proven default through catalog ordering or provider hints. +2. Connect discovery to account/config/version-scoped last-good snapshots, + TTL, a refresh lease, bounded timeout/backoff and pagination. Wire complete + drift into affected-profile revalidation and qualified fallback/block. + Incomplete discovery must preserve the incumbent and prior snapshot. +3. Normalize documented native Codex quota observations into existing typed + pools/windows, with timestamps, confidence and unknown values preserved. + Use the existing reservation fence for launch decisions and keep manual + observations available for unsupported providers. + +**Acceptance:** changed provider default does not change an approved alias; +approved promotion affects new runs only; exact pins and old runs stay fixed; +partial/auth-failed discovery does not remove models; two projects honor +fresh weekly exhaustion despite short-window availability; stale/unsupported +quota remains unknown; no permission or billing expansion occurs. Exercise +these through normal task entry, not only router/store helper calls. + +### R5 — Feed learning and trials from actual saved runs + +**Files:** terminal transition/report paths, `service.py`, `store.py`, +`learning.py`, `lifecycle.py`, migrations if needed, and public service tests. + +1. Project objective terminal facts into an idempotent final outcome record, + including failed attempts, repairs, disposition and evidence references. + Keep subjective later corrections explicit. Recover safely from a crash + between terminalization and outcome projection without duplicate records. +2. Freeze the experiment, cases, splits, profiles and budgets before either + arm launches, and run bounded paired + trials through the existing runner, account reservations and experiment + budgets. Derive consumed budget from durable attempts, not imported counters. + Carry assignment, case/split and profile provenance into outcomes. Use a + thin bounded controller, not a new general-purpose scheduler. +3. Connect public reporting, evaluation, qualification and proposal generation + to those outcomes. Trace and fix the SQLite cleanup warning without merely + suppressing it. + +**Acceptance:** a public offline run completes and appears in `report` without +manual outcome import; real public fixture runs yield evaluator-eligible +paired evidence without direct SQL state fabrication; a failed original +attempt repaired later is not credited as independently successful; late +correction, crash/replay, budgets and promotion/rollback all remain truthful. +Exercise the complete public chain: freeze experiment → run both arms → record +terminal outcomes → evaluate → qualify → promote for a new run → rerun held-out +cases → roll back. Count independent completed pairs, not labels or retries. + +### R6 — Finish the normal terminal and readiness experience + +**Files:** `cli.py`, `task_entry.py`, `diagnostics.py`, integrations, runtime +guide and public CLI tests. Use the existing service and single lead authority. + +1. Provide a supported terminal path from normal review/fix entry through + disposition without hand-written JSON, using the configured existing + headless lead or an explicit guided host path. Keep low-level JSON APIs. + Resolve omitted status/result IDs only for an unambiguous current-project + run; otherwise show choices. Supply readable output and a real next action. +2. Separate installed/registered/supported/authenticated/operation-verified + readiness. Use non-generating auth checks where supported and report + unknown otherwise. Missing auth must not advertise an unavailable workflow + as ready. Keep required logins in the user's normal provider flow. +3. Discover checks from the selected committed target and approved project + contracts. For DevSquad, include the Bash suite, relevant Python core suite + and generated-reference check. Do not guess unittest from a directory + named `tests` in an arbitrary non-Python project. Keep scope/check approval + bounded and do not let generated preparation expand permissions. + +**Acceptance:** fresh temporary install plus normal CLI review/fix reaches an +offline terminal receipt without manually authored task/decision JSON; zero, +one and multiple current-project runs behave correctly; unrelated runs are +never selected; Claude logged out is visible; target-ref check discovery is +stable; a Python defect cannot pass the DevSquad default delivery gate through +the Bash-only path. Update the quickstart to the verified commands. + +### R7 — Complete C1 within the existing runner + +Read [SELECTION-AND-COUNCIL.md](SELECTION-AND-COUNCIL.md) before this package. +It is separate from M7; its offline implementation can proceed while live +authentication gates wait. No new service, dashboard or multiple writers. + +1. Add the gated workflow/schema and `squad council` normal entry with two + distinct verified proposer identities, a critic distinct from both, and + the existing sole lead. Apply shared profile/capacity/budget rules. +2. Enforce independent proposals through stage barriers and artifact access, + then structured critique with validated labels/order and retained dissent. + Persist every attempt and support cancellation, recovery and missing quorum. +3. Run adversarial offline fixtures and the predeclared matched/held-out + comparison; obtain the required bounded subscription live receipt when + eligible identities are available. Keep automatic triggering disabled + unless its separate evidence and reviewed policy authorize it. + +**Acceptance:** no peer draft access or self-scoring; missing/invalid critic +cannot become consensus; label shuffling is correctly reversed; a wrong +majority cannot override failed checks; budgets and restart preserve history. +Inconclusive comparison is recorded as inconclusive, with automatic use off. + +### R8 — Install, prove live gates and audit closure + +1. Install the repaired immutable release and run the quickstart walkthrough, + dependency/drift/idempotence checks and required installed-SDK tests. + Independently review the repaired paths. Preserve previous releases and + saved receipts; a new release does not erase earlier failures. +2. After normal Claude login, prove the M4 real-host handoff and one bounded + installed Claude implementation → verified different-model Codex review → + mandatory checks → disposition. After Grok login, prove its supported + operation. Recheck affected Codex/Gemini surfaces when their paths change; + unchanged historical receipts are not a reason for repeated model calls. +3. Handle the classifier gate under its existing authorization: if the key + becomes available, run the prepared synthetic Jev request once, no retry, + with a $0.01 maximum and verified current pricing. Run pinned local Laya + only on the declared trigger. A larger hosted comparison needs separate + authorization. Audit every requirement and report measured limitations. + +**Acceptance:** requirement-to-evidence matrix contains exact revisions, +commands, outcomes, candidate/receipt hashes and verified scope. M5/M6/M7 and +C1 close only against their own required gates. External failures remain +explicit; keep-off is a valid classifier adoption outcome, not missing-proof +permission to claim routing improvement. + +## Verification and delivery discipline + +- First write a behavior regression that fails on the reviewed baseline, + then implement the repair. Test public transitions, not only dictionary + constructors or helper output. Keep fixture evidence distinct from live. +- Run relevant Python tests for each code slice and `bash test/run.sh` before + every commit. Run complete Python and generated-reference gates at coherent + integration boundaries. Do not repeat unaffected suites just to accumulate + counts. Use the installed SDK environment to resolve relevant optional skips. +- Cancellation tests need process inspection. If a sandbox denies it, record + that limitation and obtain the narrowly needed execution permission; do not + weaken recovery logic to make sandbox-restricted tests appear green. +- Update the milestone's review item, evidence and `RESUME.md` per coherent + checkpoint. Preserve historical receipts and report failures accurately. + End a pause with a clean committed tree; never stash. No push/merge/deploy. +- Use existing subscription harnesses for bounded required live proofs. + No credit purchase, reset redemption, paid API fallback or global AI-setting + change is authorized. Credentials and raw provider output stay outside Git. +- Report remaining work as concrete acceptance gates, not invented completion + percentages or fixed five-hour-window estimates. Shared quota depends on + actual models, context and account usage. + +## First action for Sol + +Start R1 with a failing public service regression for a check that changes +tracked source. Repair it and prove both host and headless acceptance refuse +the stale candidate. Commit that verified slice, update the R1 checkpoint, +then continue R2–R6 without waiting on the Claude login or Jev key. Retain R7 +and R8 in the full scope and continue their independent work as dependencies +become ready. diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index 265445f..dcda6c0 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -1,10 +1,12 @@ # DevSquad: coding-agent entry point **Build status: implementation in progress.** After an interruption, read -[RESUME.md](RESUME.md) first and compare it with current Git state. M1–M3 are -accepted; M4's actual Claude handoff is externally blocked and M5 is in -progress. See [backlog.json](backlog.json) for evidence. Do not restart the -architecture exercise. +[RESUME.md](RESUME.md) first and compare it with current Git state. M1/M2 are +accepted; the September 29 review reopened M3 acceptance integrity and M5–M7 +implementation gaps. M4's actual Claude handoff remains externally blocked. +Execute [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md) and use +[backlog.json](backlog.json) for evidence. Do not restart the architecture +exercise. **Full-build assignment:** Use [SOL-HANDOFF.md](SOL-HANDOFF.md) for the user's request to have Sol execute everything, test thoroughly and make normal use simple. It includes M1–M7 plus the opt-in Council feature, and adds guided task entry over the same contracts. @@ -46,9 +48,10 @@ task upload is authorized. ```text Implement DevSquad's September engineering-team plan in this repository. Read docs/plans/engineering-team/RESUME.md, current Git state, -START-HERE.md and its contracts first. Preserve existing implementation. -Start at the earliest pending milestone whose dependencies are complete. -Implement M1 and pass its gate, then continue through M2–M7 and C1. +SOL-REVIEW-FOLLOWUP.md, START-HERE.md and the relevant contracts first. +Preserve existing implementation and historical receipts. +Start with R1's candidate-integrity regression, then execute the remaining +review packages through M5–M7 and C1. Do not restart completed M1/M2 work. Preserve existing Bash 3.2 wrapper callers and their four error prefixes. Keep all distributable core files inside plugin/core; add no cloud service. Use fake CLIs for development; real provider runs are bounded smoke tests. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index e3333fe..e810e0a 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -10,9 +10,27 @@ "model_lifecycle_and_native_adapters": "MODEL-LIFECYCLE-AND-NATIVE-ADAPTERS.md", "decision_classifiers": "DECISION-CLASSIFIERS.md", "execution_brief": "SOL-HANDOFF.md", + "review_followup": "SOL-REVIEW-FOLLOWUP.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", - "next_milestone": "M7", + "next_milestone": "M3", + "review_checkpoint": { + "reviewed_revision": "f4fa6577e2e891151231c9c7d3180be6e9e23faa", + "recorded_on": "2026-09-29", + "artifact": "SOL-REVIEW-FOLLOWUP.md", + "verification": "317 Python tests completed successfully with 2 optional-SDK skips; 227 Bash assertions passed; recurring SQLite cleanup warning remains; new integrity reproductions expose missing coverage", + "assessment": "M3 acceptance integrity and M5-M7 independent engineering work are reopened. Historical receipts remain evidence for their recorded scope; credentials are not the only remaining work." + }, + "review_work_packages": [ + {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "pending", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"]}, + {"id": "R2", "title": "Observed Claude execution identity", "status": "pending", "milestones": ["M5"], "depends_on": [], "items": ["F2"]}, + {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "pending", "milestones": ["M6"], "depends_on": [], "items": ["F3"]}, + {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, + {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, + {"id": "R6", "title": "Normal terminal experience and readiness", "status": "pending", "milestones": ["M5", "M7"], "depends_on": ["R1", "R2", "R4", "R5"], "items": ["G3", "G4"]}, + {"id": "R7", "title": "Complete Council within existing runner", "status": "pending", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"]}, + {"id": "R8", "title": "Installed proofs, external gates and closure audit", "status": "pending", "milestones": ["M4", "M5", "M6", "M7", "C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": [], "note": "Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked."} + ], "milestones": [ { "id": "M1", @@ -83,8 +101,9 @@ "id": "M3", "title": "Usable branch review from terminal", "depends_on": ["M2"], - "status": "complete", + "status": "in_progress", "acceptance_section": "M3 — Ship a useful branch review", + "open_review_items": ["F1"], "evidence": [ { "kind": "implementation_checkpoint", @@ -211,8 +230,9 @@ "id": "M5", "title": "Bounded implementation with independent review", "depends_on": ["M3"], - "status": "blocked", + "status": "in_progress", "acceptance_section": "M5 — Deliver a bounded engineering change", + "open_review_items": ["F1", "F2", "G4"], "evidence": [ { "kind": "implementation_checkpoint", @@ -269,14 +289,15 @@ "availability": "tracked_tests" } ], - "blocker": "Normal Claude CLI login is required for the genuine bounded Claude implementation followed by different-model Codex review/check/disposition; offline fixtures are not substituted for the two-harness acceptance receipt" + "blocker": "Independent R1/R2/R6 repairs remain available. The later genuine Claude-to-Codex delivery gate also requires normal Claude login; fixtures cannot substitute for that receipt." }, { "id": "M6", "title": "Shared capacity and evidence-based improvement", "depends_on": ["M4", "M5"], - "status": "blocked", + "status": "in_progress", "acceptance_section": "M6 — Make capacity and improvement evidence useful", + "open_review_items": ["F3", "F4", "G1", "G2"], "decision_helper_work_packages": [ { "id": "M6-D1", @@ -381,14 +402,15 @@ "availability": "tracked_fixture_and_tests" } ], - "blocker": "All independent M6 implementation is complete; the single capped Jev M6-D2 measurement requires TYPESAFE_API_KEY, and M6-D3 runs only if its declared Jev fallback trigger fires" + "blocker": "Independent R3-R5 evidence, routing, learning and observation integration remains. Only the separate single-request Jev measurement requires TYPESAFE_API_KEY; Laya remains conditional." }, { "id": "M7", "title": "Installation, migration and all-surface proof", "depends_on": ["M6"], - "status": "blocked", + "status": "in_progress", "acceptance_section": "M7 — Package, migrate and prove every requested surface", + "open_review_items": ["F4", "G3", "G4"], "evidence": [ { "kind": "packaging_checkpoint", @@ -427,7 +449,7 @@ "availability": "portable_redacted" } ], - "blocker": "All independently executable M7 work is complete. Universal surface completion requires normal Claude login for the M4/M5 handoff and installed delivery receipts plus renewed Grok authentication for its live smoke." + "blocker": "Independent R4/R6 normal-entry, readiness and check-discovery work remains. Final live surface and delivery receipts additionally require Claude login and renewed Grok authentication." } ], "extensions": [ From 6848f1156bee10e1ef2617402dfc7eaebdc58076 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 08:20:33 -0700 Subject: [PATCH 132/197] WIP checkpoint: R1 candidate integrity repair with public regressions and evidence (2026-09-29 08:20) --- docs/generated/core-reference.md | 4 +- docs/plans/engineering-team/CONTRACTS.md | 20 ++ docs/plans/engineering-team/M3-STATUS.md | 12 +- docs/plans/engineering-team/M5-STATUS.md | 7 +- docs/plans/engineering-team/RESUME.md | 32 +- .../engineering-team/SOL-REVIEW-FOLLOWUP.md | 24 +- docs/plans/engineering-team/START-HERE.md | 2 +- docs/plans/engineering-team/backlog.json | 11 +- .../R1-candidate-integrity-2026-09-29.json | 43 +++ plugin/core/schemas/check-result.schema.json | 33 +- plugin/core/schemas/task.schema.json | 2 +- plugin/core/src/devsquad/reports.py | 35 +- plugin/core/src/devsquad/review_worker.py | 87 ++++- plugin/core/src/devsquad/service.py | 8 +- plugin/core/src/devsquad/validation.py | 11 +- plugin/core/src/devsquad/workflows.py | 86 ++++- plugin/core/src/devsquad/workspaces.py | 66 +++- test/core/test_check_integrity_runtime.py | 331 ++++++++++++++++++ test/core/test_review_runtime.py | 1 + test/core/test_review_workflow.py | 31 ++ test/core/test_validation.py | 11 + test/core/test_workspaces.py | 57 +++ 22 files changed, 848 insertions(+), 66 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R1-candidate-integrity-2026-09-29.json create mode 100644 test/core/test_check_integrity_runtime.py diff --git a/docs/generated/core-reference.md b/docs/generated/core-reference.md index d966310..617266f 100644 --- a/docs/generated/core-reference.md +++ b/docs/generated/core-reference.md @@ -62,7 +62,7 @@ runtime never infers them from chat history. | File | Identifier | Required top-level fields | SHA-256 | |---|---|---|---| | `adapter.schema.json` | `https://devsquad.local/schemas/adapter-v1.json` | `schema_version`, `name`, `transport`, `binary_candidates`, `capabilities`, `permission_profiles` | `fc6b811b59bd2920d3dc1588d5bf447515a25d87219d07d76e1764efd60ad9c0` | -| `check-result.schema.json` | `https://devsquad.local/schemas/check-result-v1.json` | `schema_version`, `candidate_sha256`, `target_oid`, `id`, `argv`, `cwd`, `required_to_pass`, `status`, `returncode`, `error_code`, `duration_ms`, `stdout`, `stderr` | `6597b7e6839d6b215c1bb692c3ee5db47ee88495d96bdfa4730db6f56b2c0c32` | +| `check-result.schema.json` | `https://devsquad.local/schemas/check-result-v2.json` | `schema_version`, `candidate_sha256`, `target_oid`, `id`, `argv`, `cwd`, `required_to_pass`, `status`, `returncode`, `error_code`, `duration_ms`, `stdout`, `stderr` | `5145755d8e60503151f8938f1753c657724f1a2de3a442500c14eda158fe04a1` | | `execution-identity.schema.json` | `https://devsquad.local/schemas/execution-identity-v1.json` | `harness`, `harness_version`, `model_provider`, `model_family`, `model`, `effort`, `tools`, `permissions`, `account_pool`, `verification` | `99386fb81c3567bdd3c880b37d363d5ca61cccb3ff35add5eb01a1ad916dfa90` | | `launch-spec.schema.json` | `https://devsquad.local/schemas/launch-spec-v1.json` | `schema_version`, `adapter`, `transport`, `argv`, `cwd`, `stdin_path`, `timeout_seconds`, `requested`, `environment` | `12e2c3ffbfee6b3a9e93e41d9dd7d6a2c82a9d62e981d9378307b12f6ff12f5c` | | `normalized-result.schema.json` | `https://devsquad.local/schemas/normalized-result-v1.json` | `schema_version`, `execution_status`, `error_code`, `output`, `artifact_status`, `acceptance_status`, `requested`, `observed`, `native_ids`, `events` | `8fb7eec0d3cfc952b9fa839b5fdbe950d5cfe06ed69bbbd3bcf835aad9df753d` | @@ -70,7 +70,7 @@ runtime never infers them from chat history. | `profile.schema.json` | `https://devsquad.local/schemas/profile-v1.json` | `id`, `harness`, `model_family`, `model_id`, `effort`, `required_tools`, `permission_policy`, `account_pool_id`, `billing_mode`, `quality_status`, `evidence_refs` | `fc51b49cd7dd6a134eedf2c2c941301392aef29a2163ac2d307b4fb78509ec96` | | `profiles.schema.json` | `https://devsquad.local/schemas/profiles-v1.json` | `schema_version`, `profiles`, `bindings` | `c14823c89d1b81ee93b74f5bbc2701ae975825db795b5f0937c28fc057d84cf1` | | `review-result.schema.json` | `https://devsquad.local/schemas/review-result-v1.json` | `schema_version`, `candidate_sha256`, `base_oid`, `target_oid`, `review_mode`, `verdict`, `summary`, `findings` | `96b852543f92dca21dafd6c0bc954f8f56d3dbfb0981d601cc05b9c1c9ba6e6b` | -| `task.schema.json` | `https://devsquad.local/schemas/task-v1.json` | `schema_version`, `project`, `workflow`, `goal`, `task_class`, `acceptance`, `checks`, `scope`, `lead`, `routing`, `budget`, `origin` | `ec2a700b9dac653b49ffe0bfc6fa24bda92a2a144b0c94e9b66307d93c3bbc19` | +| `task.schema.json` | `https://devsquad.local/schemas/task-v1.json` | `schema_version`, `project`, `workflow`, `goal`, `task_class`, `acceptance`, `checks`, `scope`, `lead`, `routing`, `budget`, `origin` | `afe02492bccc098c444be6095150b9f9af97d1dbda6187c69bff6b8bd0c4eaac` | ## Task-shape example diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index 35d29c2..9fcb2a3 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -167,6 +167,26 @@ For `issue-delivery`, `revise` loops back to implementation. For `branch-review` Bind artifacts, findings, tests, dispositions and criteria to a candidate tree/patch hash and resolved baseline. Freeze an implementation candidate into a local run-owned commit before review; reject changes outside scope and capture untracked permitted outputs intentionally. Run checks in a separate candidate worktree so test-generated files cannot alter the reviewed candidate. If a check must change source, that creates a new candidate needing new review. A subsequent code change invalidates affected evidence and triggers review/checks again. Run results contain a patch/commit reference and integration instructions; the coordinator does not alter the user's current checkout. +Each check may declare `output_paths` (default `[]`): at most 32 canonical +repository-relative file/directory paths for untracked build artifacts. These +are frozen with the check plan; root/Git metadata/traversal paths are forbidden. +An output declaration never permits changing tracked candidate files. Ignored +files are not implicitly approved outputs. Approved outputs may be consumed by +later checks; undeclared new inputs, tracked content/mode/index changes, HEAD +changes, or mutation of the review worktree invalidate the check and skip the +remaining checks. This integrity gate also blocks report-only check acceptance. +Checks compare actual tracked bytes, independent of Git stat-cache and +assume-unchanged hints. Unsupported Git submodule inputs fail closed. + +Check evidence v2 records integrity status, bounded mutation path/digest +witnesses and complete state fingerprints alongside the original process exit +and output. Both handoff and terminal reports retain the failure, including +when the lead fails. Legacy v1 receipts remain readable and completed decisions +remain replayable; they cannot authorize a new acceptance. Start a new run to +refresh unverified legacy checks. Boundary verification applies to trusted +checks; it is not a sandbox against a malicious command that mutates and +restores source entirely within one check invocation. + Every attempt receipt includes role, parent step, profile/policy/prompt/schema versions, requested/observed model/effort, tools/permissions, runtime version, input/output hashes, process verdict, artifact verdict, latency, usage by source and error. The run receipt also includes criteria results, all attempt IDs, fallbacks, revisions, lead repairs, final disposition and remaining limitations. Host work is recorded as externally observed with unknown usage where unmeasured; do not omit it or estimate it as zero. ## 6. Capacity without false precision diff --git a/docs/plans/engineering-team/M3-STATUS.md b/docs/plans/engineering-team/M3-STATUS.md index 8a7d0f6..7bac96d 100644 --- a/docs/plans/engineering-team/M3-STATUS.md +++ b/docs/plans/engineering-team/M3-STATUS.md @@ -1,9 +1,11 @@ # M3 implementation status -M3 is **reopened for candidate-integrity repair R1** after the September 29 -review at `f4fa657`. A check that changes tracked source can produce passing -evidence for the original candidate. See F1 and its required public-service -regression in [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). +M3's reopened **candidate-integrity repair R1 is verified in source**. Public +review/delivery regressions now block source-changing and undeclared-input +checks through host/headless disposition, retain evidence and skip dependent +checks. The final core gate ran 330 tests successfully (two optional SDK skips). +See [R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json). Installed +refresh/revalidation remains R8 work; the old installed payload is unchanged. M3 was accepted at implementation checkpoint `1737667`. That historical gate passes 188 core tests with `ResourceWarning` promoted to an error and all 10 @@ -16,7 +18,7 @@ additional provider turn was used for the closeout. | Deterministic selection | Versioned profile aliases, exact pins, explicit `none`/`policy` fallback, permission/billing filters and typed capacity produce stable frozen routing snapshots | verified offline | | Frozen branch input | Base/target refs resolve to exact OIDs; committed config hashes, candidate hash and detached review/check workspaces remain stable while the submitted checkout, index and HEAD are preserved | verified offline | | Reviewer evidence | Strict ordinary/adversarial prompts, review/check/evaluation schemas, candidate binding, read-only identity checks and malformed/denied/disconnected output faults prevent unsupported success | verified offline | -| Check and lead gates | Report-only/required failures and bounded dispositions have historical tests; source-changing checks expose a candidate-integrity gap | reopened: R1 | +| Check and lead gates | Source/HEAD/index/mode/undeclared-input mutation invalidates evidence even for report-only checks; host/headless refuse acceptance and later checks are skipped | R1 verified offline; installed refresh pending R8 | | Durable host handoff | Waiting JSON/Markdown packets, claim leases, renewal/takeover, stale completion fencing, replay and crash-resume continuation share the saved run ledger | verified offline | | Terminal reporting | Success, rejection, preflight failure, worker failure, cancellation, waiting cancellation, timeout and budget exhaustion publish receipt JSON/Markdown, events, manifest and result receipt | verified offline | | Headless leadership | Offline and native headless leads run as separate fenced attempts, verify their own frozen identity/evidence and terminalize without host intervention | verified offline | diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index 6610cbf..271618b 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -2,7 +2,8 @@ M5 is **in progress with independent repairs available**. The September 29 review at `f4fa657` reopened candidate integrity (F1), observed Claude identity -(F2) and complete normal-command check coverage (G4). Execute R1/R2/R6 in +(F2) and complete normal-command check coverage (G4). F1/R1 is now repaired +and verified offline; execute R2/R6 in [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The genuine Claude-to-Codex live gate additionally remains blocked on normal Claude login. @@ -12,9 +13,9 @@ live gate additionally remains blocked on normal Claude login. | Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | verified offline at `0e88d73` | | Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `0e88d73` | | Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | reopened under R2; prior fixtures do not detect fabricated observed identity; live proof pending | -| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | reopened under R1/R6 for check mutation and normal-command check coverage | +| Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | R1 check-mutation repair verified offline; R6 normal-command check coverage remains | | Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | correction/budgets verified at `4c76887`; same-permission delivery fallback verified at `b7d90cc` | -| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | historical failure gates pass; R1/R2 must reject invalid candidate/identity evidence | +| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | R1 candidate integrity now enforced; R2 identity repair remains | | Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | success, repair, fallback, failure and cancellation history verified offline at `b7d90cc` | | Crash recovery | Killing a live implementation supervisor cannot create a duplicate writer on resume | prelaunch and live revised-writer recovery verified offline at `f199cd2` | | Live acceptance | One bounded issue completes across at least two authenticated subscription harnesses with different verified models | pending repairs and normal Claude CLI login | diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 0fc8c9e..18c798a 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,15 +6,27 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Review correction and next action +R1's source repair is verified after `399d93d`: the public regression first +reproduced four unsafe mutation paths (review/delivery × host/headless). The +repair adds check-boundary fingerprints, explicit permitted output paths, +non-overridable invalidation, durable mutation evidence and historical receipt +compatibility. The final core gate ran **330 tests successfully, with 2 optional +SDK skips**. The final independent bounded R1 audit found no remaining actionable +issues; all **227 Bash assertions** and the generated-reference check passed. +See [R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json) for the +full verification record, source fingerprints and limitations. The installed +runtime is not yet refreshed; that remains R8 work. + The review of `f4fa6577e2e891151231c9c7d3180be6e9e23faa` supersedes the earlier -claim that only credentials remain. M1/M2 remain accepted; M3 is reopened for -candidate integrity (F1); M4 retains its real-Claude-host gate; M5–M7 have +claim that only credentials remain. M1/M2 remain accepted; M3's F1 source repair +is verified with installed refresh pending; M4 retains its real-Claude-host gate; M5–M7 have independent repairs and integration work; C1 remains pending full-delivery -scope. No runtime repairs were made during the review or this planning update. +scope. R1 is the first implemented repair; do not confuse it with full closure. -Execute [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), starting at **R1**: -reproduce source-mutating trusted checks through the public service and prevent -acceptance of invalid candidate evidence. Then repair Claude identity (R2), +Execute [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), starting at **R2**: +repair Claude identity using native `modelUsage`, keeping effective effort +unknown when unreported. The plan contains the bounded local/primary-source +investigation; no live Claude generation was used. Continue with independent experiment evidence (R3), routing/catalog/quota (R4), learning runtime connections (R5), and normal terminal/readiness/check discovery (R6). R7 covers C1 and R8 covers installed/live closure. Authentication and the Jev @@ -302,10 +314,10 @@ provider paths must not be advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Execute R1's failing public-service candidate-integrity regression, repair - the shared review/delivery gate and checkpoint its evidence. Continue R2/R3, - then R4–R6 in dependency order. Preserve later implementation and historical - receipts; do not rewrite the architecture or reset completed work. +2. Execute R2's offline Claude model-identity regressions and strict import gate, + then R3 and R4–R6 in dependency order. Preserve the verified R1 repair, + explicit check `output_paths` contract and historical receipts. Do not + rewrite the architecture or reset completed work. 3. Keep Claude and Grok live probes paused until normal login is restored. Afterward, run the M4 Claude handoff, the M5 installed Claude-to-Codex delivery and one bounded Grok operation, retaining only redacted evidence. diff --git a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md index 2626157..d797410 100644 --- a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md +++ b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md @@ -20,6 +20,12 @@ Read `AGENTS.md`, `CONTRIBUTING.md`, [RESUME.md](RESUME.md), this document and for each work package. Do not reload the historical chat or restart M1/M2. This review supersedes the earlier claim that only credentials remain. +Current execution checkpoint: **R1 source repair verified**, with public +mutation regressions, a ten-case workspace mutation matrix, historical replay +coverage and 330-test core gate (two optional-SDK skips). See +[R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json). Start at **R2** +after checking current Git state. R2–R8 and installed refresh remain open. + ## Feedback on what has been built Preserve the substantial working implementation: the durable runner and @@ -103,6 +109,17 @@ restore invalidated evidence. Preserve per-check temporary HOME behavior. **Files:** `claude_delivery_worker.py`, `workflows.py`, adapter fixtures and delivery tests. Implement conformance offline before spending a live turn. +Investigation checkpoint (September 29): installed Claude `2.1.220` exposes +`session_id` and `modelUsage` in native result JSON. Preserve all reported model +entries; `canonicalModel` is pricing normalization, not serving-model proof. +The result does not report effective effort, so keep it unknown rather than +copying `--effort`. Multiple entries do not identify a unique writer. Pin the +parser to tested local capability; current online docs can describe newer CLI +features. Primary references: [result schema](https://code.claude.com/docs/en/agent-sdk/python#resultmessage), +[model aliases and effort limits](https://code.claude.com/docs/en/model-config), +and [usage attribution](https://code.claude.com/docs/en/agent-sdk/cost-tracking). +Do not replace subscription-compatible safe mode with `--bare`. + 1. Verify the supported CLI's native model/session/usage fields against its installed schema/help and current official documentation when necessary. Record requested settings separately from actual reported identity. @@ -294,9 +311,8 @@ permission to claim routing improvement. ## First action for Sol -Start R1 with a failing public service regression for a check that changes -tracked source. Repair it and prove both host and headless acceptance refuse -the stale candidate. Commit that verified slice, update the R1 checkpoint, -then continue R2–R6 without waiting on the Claude login or Jev key. Retain R7 +R1's source repair is verified; continue at R2 with failing native-identity +and import-tampering regressions. Preserve R1's check-output contract and +mutation evidence. Continue R2–R6 without waiting on the Claude login or Jev key. Retain R7 and R8 in the full scope and continue their independent work as dependencies become ready. diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index dcda6c0..580999a 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -50,7 +50,7 @@ Implement DevSquad's September engineering-team plan in this repository. Read docs/plans/engineering-team/RESUME.md, current Git state, SOL-REVIEW-FOLLOWUP.md, START-HERE.md and the relevant contracts first. Preserve existing implementation and historical receipts. -Start with R1's candidate-integrity regression, then execute the remaining +R1's source repair is verified; start with R2's observed-identity regressions, then execute the remaining review packages through M5–M7 and C1. Do not restart completed M1/M2 work. Preserve existing Bash 3.2 wrapper callers and their four error prefixes. Keep all distributable core files inside plugin/core; add no cloud service. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index e810e0a..198284d 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -13,7 +13,7 @@ "review_followup": "SOL-REVIEW-FOLLOWUP.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", - "next_milestone": "M3", + "next_milestone": "M5", "review_checkpoint": { "reviewed_revision": "f4fa6577e2e891151231c9c7d3180be6e9e23faa", "recorded_on": "2026-09-29", @@ -22,7 +22,7 @@ "assessment": "M3 acceptance integrity and M5-M7 independent engineering work are reopened. Historical receipts remain evidence for their recorded scope; credentials are not the only remaining work." }, "review_work_packages": [ - {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "pending", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"]}, + {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, {"id": "R2", "title": "Observed Claude execution identity", "status": "pending", "milestones": ["M5"], "depends_on": [], "items": ["F2"]}, {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "pending", "milestones": ["M6"], "depends_on": [], "items": ["F3"]}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, @@ -103,7 +103,8 @@ "depends_on": ["M2"], "status": "in_progress", "acceptance_section": "M3 — Ship a useful branch review", - "open_review_items": ["F1"], + "open_review_items": [], + "review_resolution": "F1 source repair is verified under R1; affected installed-runtime refresh remains R8, so M3 is not reclosed yet.", "evidence": [ { "kind": "implementation_checkpoint", @@ -232,7 +233,7 @@ "depends_on": ["M3"], "status": "in_progress", "acceptance_section": "M5 — Deliver a bounded engineering change", - "open_review_items": ["F1", "F2", "G4"], + "open_review_items": ["F2", "G4"], "evidence": [ { "kind": "implementation_checkpoint", @@ -289,7 +290,7 @@ "availability": "tracked_tests" } ], - "blocker": "Independent R1/R2/R6 repairs remain available. The later genuine Claude-to-Codex delivery gate also requires normal Claude login; fixtures cannot substitute for that receipt." + "blocker": "R1 source repair is verified; independent R2/R6 repairs remain available. The later genuine Claude-to-Codex delivery gate also requires normal Claude login; fixtures cannot substitute for that receipt." }, { "id": "M6", diff --git a/docs/plans/engineering-team/evidence/R1-candidate-integrity-2026-09-29.json b/docs/plans/engineering-team/evidence/R1-candidate-integrity-2026-09-29.json new file mode 100644 index 0000000..2c4846f --- /dev/null +++ b/docs/plans/engineering-team/evidence/R1-candidate-integrity-2026-09-29.json @@ -0,0 +1,43 @@ +{ + "schema_version": 1, + "work_package": "R1", + "status": "source_verified", + "baseline_revision": "399d93d", + "recorded_on": "2026-09-29", + "scope": "Shared branch-review and issue-delivery candidate integrity through trusted checks", + "baseline_reproduction": { + "command": "python3 -m unittest discover -s test/core -p test_check_integrity_runtime.py -v", + "result": "Four mutation regressions failed on unmodified production; two permitted-build-output positive cases passed. Both host and headless paths could continue checking altered source." + }, + "acceptance_mapping": [ + {"requirement": "Public review/delivery and host/headless cannot accept source-changing report-only checks", "test": "test_check_integrity_runtime.PublicCheckIntegrityTest"}, + {"requirement": "Undeclared new source is rejected; approved untracked build artifacts remain usable", "test": "test_check_integrity_runtime.PublicCheckIntegrityTest"}, + {"requirement": "Later checks never consume a contaminated tree; exact candidate and original checkout remain unchanged", "test": "test_check_integrity_runtime.PublicCheckIntegrityTest"}, + {"requirement": "Restart/replay preserves invalidation and terminal receipt evidence", "test": "test_check_integrity_runtime.PublicCheckIntegrityTest"}, + {"requirement": "Delete, executable mode, staged/index-only edits, assume-unchanged, HEAD checkout/attachment, symlinks, ignored source and review-tree mutation are detected", "test": "test_workspaces.ReviewWorkspaceTest.test_check_integrity_covers_tracked_modes_index_head_and_hidden_changes"}, + {"requirement": "Legacy evidence stays readable but cannot authorize a new acceptance; completed submissions remain replayable", "test": "test_review_workflow.BranchReviewWorkflowTest.test_legacy_checks_remain_readable_but_cannot_be_imported_or_accepted_anew and test_check_integrity_runtime.PublicCheckIntegrityTest.test_terminal_completion_replay_preserves_historical_acceptance"}, + {"requirement": "Output declarations and v2 integrity witnesses cannot widen frozen checks", "test": "test_validation.AdversarialValidationTest.test_check_outputs_are_explicit_bounded_relative_paths and test_review_workflow.BranchReviewWorkflowTest.test_integrity_is_non_overridable_and_strictly_bound_to_frozen_outputs"}, + {"requirement": "Per-check temporary HOME remains isolated", "test": "test_review_runtime.DurableBranchReviewTest.test_each_trusted_check_gets_a_fresh_isolated_home"} + ], + "verification": { + "core": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -v: 330 tests ran in 190.386 seconds; OK with 2 optional-SDK skips. The existing SQLite cleanup warning remains recorded.", + "bash": "bash test/run.sh: 11 files, 227 assertions passed before the R1 checkpoint; no real provider CLIs required.", + "generated_reference": "Current after packaged task/check-result schema updates", + "independent_review": "Bounded leaf review identified undeclared source and legacy replay gaps, both repaired; final read-only R1 audit reported no further actionable issues.", + "earlier_runs": "An intermediate 71-test run overlapped schema edits and is not verification. The first stable 330-test gate found two path-assertion mistakes and one existing fixture needing an explicit generated-output declaration; none is counted as a pass." + }, + "source_sha256": { + "plugin/core/src/devsquad/review_worker.py": "883717bb8e72ed714e5f3362bb9649b549b6ab3ea9568c47a10087278a63ca55", + "plugin/core/src/devsquad/workspaces.py": "5b045ee74935ee69eb27f93c959a6cb8a5b3e366a1ea146bd75b3fb65eeedfe0", + "plugin/core/src/devsquad/workflows.py": "449237aa230e6a778be42f21f931106beeb15326f1a4fd5054b99fce5e47b7dc", + "plugin/core/src/devsquad/service.py": "1f019f4603c14098a93984db1a7d5c48083081687b0461936bdbed9aa5b2a3ae", + "plugin/core/src/devsquad/reports.py": "11299daaf9a33e20fbb2bbda1983ed730702b67d1292be330d5e8f0b33228ec4", + "plugin/core/src/devsquad/validation.py": "0ed79e40c762d3ab609b5553296135620529fd874f5cc8a5c8c83ef8e62cd5b2" + }, + "limitations": [ + "No real provider generation or installed runtime replacement occurred; refreshed installed/live proof belongs to R8.", + "Integrity is checked at trusted command boundaries, not as a sandbox against deliberate mutate-and-restore inside one invocation.", + "Git submodule candidate inputs fail closed pending explicit support.", + "The existing unclosed SQLite ResourceWarning remains assigned to R5." + ] +} diff --git a/plugin/core/schemas/check-result.schema.json b/plugin/core/schemas/check-result.schema.json index ab09628..7cfbab5 100644 --- a/plugin/core/schemas/check-result.schema.json +++ b/plugin/core/schemas/check-result.schema.json @@ -1,24 +1,51 @@ { "$schema": "https://json-schema.org/draft/2020-12/schema", - "$id": "https://devsquad.local/schemas/check-result-v1.json", + "$id": "https://devsquad.local/schemas/check-result-v2.json", "type": "object", "additionalProperties": false, "required": ["schema_version", "candidate_sha256", "target_oid", "id", "argv", "cwd", "required_to_pass", "status", "returncode", "error_code", "duration_ms", "stdout", "stderr"], "properties": { - "schema_version": {"const": 1}, + "schema_version": {"enum": [1, 2]}, "candidate_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, "target_oid": {"type": "string", "pattern": "^[0-9a-f]{40}$"}, "id": {"type": "string", "minLength": 1}, "argv": {"type": "array", "minItems": 1, "items": {"type": "string", "minLength": 1}}, "cwd": {"type": "string"}, "required_to_pass": {"type": "boolean"}, - "status": {"enum": ["passed", "failed", "timed_out", "launch_failed"]}, + "status": {"enum": ["passed", "failed", "timed_out", "launch_failed", "invalidated", "not_run"]}, + "output_paths": {"type": "array", "maxItems": 32, "uniqueItems": true, "items": {"type": "string", "minLength": 1}}, + "integrity": { + "type": "object", "additionalProperties": false, + "required": ["status", "reasons", "before_state_sha256", "after_state_sha256", "changes", "changes_truncated"], + "properties": { + "status": {"enum": ["verified", "violated", "not_run"]}, + "reasons": {"type": "array", "maxItems": 20, "items": {"type": "string", "minLength": 1, "maxLength": 200}}, + "before_state_sha256": {"type": ["string", "null"], "pattern": "^[0-9a-f]{64}$"}, + "after_state_sha256": {"type": ["string", "null"], "pattern": "^[0-9a-f]{64}$"}, + "changes_truncated": {"type": "boolean"}, + "changes": {"type": "array", "maxItems": 100, "items": { + "type": "object", "additionalProperties": false, + "required": ["workspace", "path", "before", "after"], + "properties": { + "workspace": {"enum": ["check", "review"]}, + "path": {"type": "string", "minLength": 1, "maxLength": 4096}, + "before": {"type": ["string", "null"], "pattern": "^([0-9a-f]{64}|undeclared)$"}, + "after": {"type": ["string", "null"], "pattern": "^([0-9a-f]{64}|undeclared)$"} + } + }} + } + }, "returncode": {"type": ["integer", "null"]}, "error_code": {"enum": [null, "TIMEOUT", "CLI_ERROR"]}, "duration_ms": {"type": "integer", "minimum": 0}, "stdout": {"$ref": "#/$defs/stream"}, "stderr": {"$ref": "#/$defs/stream"} }, + "allOf": [ + {"if": {"properties": {"schema_version": {"const": 2}}}, + "then": {"required": ["integrity", "output_paths"]}, + "else": {"not": {"anyOf": [{"required": ["integrity"]}, {"required": ["output_paths"]}]}, "properties": {"status": {"enum": ["passed", "failed", "timed_out", "launch_failed"]}}}} + ], "$defs": { "stream": { "type": "object", diff --git a/plugin/core/schemas/task.schema.json b/plugin/core/schemas/task.schema.json index 4413c6c..b779769 100644 --- a/plugin/core/schemas/task.schema.json +++ b/plugin/core/schemas/task.schema.json @@ -13,7 +13,7 @@ "$defs": { "project": {"type":"object","additionalProperties":false,"required":["repo_path","base_ref","target_ref"],"properties":{"repo_path":{"type":"string","pattern":"^/"},"base_ref":{"type":"string","minLength":1},"target_ref":{"type":"string","minLength":1}}}, "acceptance": {"type":"object","additionalProperties":false,"required":["id","description","evidence_kind"],"properties":{"id":{"type":"string","minLength":1},"description":{"type":"string","minLength":1},"evidence_kind":{"enum":["review","check","artifact","host"]}}}, - "check": {"type":"object","additionalProperties":false,"required":["id","argv","cwd","timeout_seconds","required_to_pass"],"properties":{"id":{"type":"string","minLength":1},"argv":{"type":"array","minItems":1,"items":{"type":"string","minLength":1}},"cwd":{"type":"string"},"timeout_seconds":{"type":"integer","minimum":1},"required_to_pass":{"type":"boolean"}}}, + "check": {"type":"object","additionalProperties":false,"required":["id","argv","cwd","timeout_seconds","required_to_pass"],"properties":{"id":{"type":"string","minLength":1},"argv":{"type":"array","minItems":1,"items":{"type":"string","minLength":1}},"cwd":{"type":"string"},"timeout_seconds":{"type":"integer","minimum":1},"required_to_pass":{"type":"boolean"},"output_paths":{"type":"array","maxItems":32,"uniqueItems":true,"items":{"type":"string","minLength":1}}}}, "scope": {"type":"object","additionalProperties":false,"required":["read_paths","write_paths"],"properties":{"read_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}},"write_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}}}}, "lead": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["host","headless"]}}}, "override": {"type":"object","additionalProperties":false,"required":["profile_id"],"properties":{"profile_id":{"type":"string","minLength":1},"fallback":{"enum":["none","policy"]}}}, diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index 355f111..2e29188 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -186,6 +186,7 @@ def build_handoff_reports( ) else: lines.append("- No supported findings were reported.") + lines.extend(_check_markdown(packet.get("checks", []))) lines.extend(["", "## Instructions", "", str(packet.get("instructions") or "")]) return { json_name: (canonical_json(report) + "\n").encode(), @@ -282,6 +283,13 @@ def build_early_terminal_reports( "status": "not_evaluated", "evidence_refs": [], } for criterion in task.get("acceptance", []) if isinstance(criterion, dict)] + # A failed lead does not erase the current candidate's completed review or + # check integrity verdict. Never project a prior delivery candidate here. + latest_review = next((item for item in reversed(prior) + if item.get("role") == "reviewer" + and isinstance(item.get("evaluation"), dict) + and item["evaluation"].get("candidate_sha256") == workspace.get("candidate_sha256") + ), {}) receipt = { "schema_version": 1, "run_id": run_id, @@ -299,9 +307,9 @@ def build_early_terminal_reports( ), }, "routing": frozen.get("routing"), - "review": None, - "checks": [], - "evaluation": None, + "review": latest_review.get("review"), + "checks": latest_review.get("checks", []), + "evaluation": latest_review.get("evaluation"), "criteria": criteria, "attempts": prior + ([attempt_projection] if attempt_projection else []), "dispositions": dispositions, @@ -366,6 +374,7 @@ def build_early_terminal_reports( "Start a new run with a new idempotency key after correcting the recorded error.", "", ] + lines.extend(_check_markdown(receipt["checks"])) return _contents_with_manifest( run_id, receipt, "\n".join(lines), events_content, projected, ) @@ -496,6 +505,18 @@ def project_branch_review_history( return _history(entries, snapshot) +def _check_markdown(checks: list[dict[str, Any]]) -> list[str]: + lines = ["", "## Checks", ""] + for check in checks: + requirement = "required" if check["required_to_pass"] else "report-only" + lines.append(f"- `{check['id']}`: **{check['status']}** ({requirement})") + for reason in check.get("integrity", {}).get("reasons", []): + lines.append(f" - Candidate integrity: `{reason}`") + if not checks: + lines.append("- No checks were declared.") + return lines + + def _markdown(receipt: dict[str, Any]) -> str: review = receipt["review"] lines = [ @@ -525,13 +546,7 @@ def _markdown(receipt: dict[str, Any]) -> str: f" Evidence: {finding['evidence']}", ]) lines.append("") - lines.extend(["## Checks", ""]) - if receipt["checks"]: - for check in receipt["checks"]: - requirement = "required" if check["required_to_pass"] else "report-only" - lines.append(f"- `{check['id']}`: **{check['status']}** ({requirement})") - else: - lines.append("- No checks were declared.") + lines.extend(_check_markdown(receipt["checks"])) lines.extend([ "", "## Lead disposition", diff --git a/plugin/core/src/devsquad/review_worker.py b/plugin/core/src/devsquad/review_worker.py index 297bc1c..6911f85 100644 --- a/plugin/core/src/devsquad/review_worker.py +++ b/plugin/core/src/devsquad/review_worker.py @@ -20,7 +20,7 @@ make_branch_review_evidence, validate_review_document, ) -from .workspaces import dirty_paths, reset_check_workspace +from .workspaces import candidate_input_state, dirty_paths, reset_check_workspace MAX_SNAPSHOT_BYTES = 2 * 1024 * 1024 @@ -166,15 +166,82 @@ def run_review_and_checks( _verify_finding_locations(normalized_review, review_root) if dirty_paths(review_root): raise ContractError("reviewer modified the frozen read-only workspace") - checks = [ - _run_check( - check, - checks_root, - workspace["candidate_sha256"], - workspace["target_oid"], - ) - for check in task["checks"] - ] + target_oid = workspace["target_oid"] + roots = {"check": checks_root, "review": review_root} + outputs = [path for check in task["checks"] for path in check.get("output_paths", [])] + baseline = { + label: candidate_input_state(root, target_oid, outputs if label == "check" else ()) + for label, root in roots.items() + } + + baseline_sha256 = hashlib.sha256(canonical_json(baseline).encode()).hexdigest() + + def inspect_integrity() -> dict[str, Any]: + reasons, changes, states = [], [], {} + for label, root in roots.items(): + try: + current = candidate_input_state(root, target_oid, outputs if label == "check" else ()) + except (ContractError, OSError, ValueError): + reasons.append(f"{label}:inspection_failed") + states[label] = None + continue + states[label] = current + reasons.extend( + f"{label}:{field}_changed" for field, value in baseline[label].items() + if field != "paths" and current[field] != value + ) + before, after = baseline[label]["paths"], current["paths"] + for path in sorted(before.keys() | after.keys()): + if before.get(path) != after.get(path): + changes.append({ + "workspace": label, "path": path, + "before": before.get(path), "after": after.get(path), + }) + return { + "reasons": reasons, + "before_state_sha256": baseline_sha256, + "after_state_sha256": hashlib.sha256(canonical_json(states).encode()).hexdigest(), + "changes": changes[:100], "changes_truncated": len(changes) > 100, + } + + checks = [] + invalidated = False + for check in task["checks"]: + inspection = inspect_integrity() if not invalidated else { + "reasons": [], "before_state_sha256": None, "after_state_sha256": None, + "changes": [], "changes_truncated": False, + } + reasons = inspection["reasons"] + if invalidated or reasons: + # Keep one outcome for every declared check without executing any + # dependent command on a contaminated candidate. + result = { + "candidate_sha256": workspace["candidate_sha256"], + "target_oid": target_oid, + **{key: check[key] for key in ("id", "argv", "cwd", "required_to_pass")}, + "status": "not_run" if invalidated else "invalidated", + "returncode": None, + "error_code": "CLI_ERROR", + "duration_ms": 0, + "stdout": _empty_stream(), + "stderr": _empty_stream(), + } + integrity_status = "not_run" if invalidated else "violated" + reasons = reasons or ["prior_check_invalidated_candidate"] + else: + result = _run_check( + check, checks_root, workspace["candidate_sha256"], target_oid, + ) + inspection = inspect_integrity() + reasons = inspection["reasons"] + integrity_status = "violated" if reasons else "verified" + if reasons: + result.update(status="invalidated", error_code="CLI_ERROR") + result["schema_version"] = 2 + result["output_paths"] = check.get("output_paths", []) + result["integrity"] = {**inspection, "status": integrity_status, "reasons": reasons} + invalidated = invalidated or integrity_status != "verified" + checks.append(result) return make_branch_review_evidence( snapshot, normalized_review, checks, **attempt_metadata, ) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 1f5c20a..b086825 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -47,6 +47,7 @@ from .validation import validate_task from .workflows import ( apply_lead_disposition, + require_check_integrity, decode_headless_lead_evidence, review_mode, validate_branch_review_handoff, @@ -1352,6 +1353,8 @@ def _review_gate( ) -> tuple[dict[str, Any], dict[str, Any]]: packet = validate_saved_review_handoff(handoff.packet, snapshot) validate_handoff_decision_evidence(decision, packet) + if decision.get("disposition") == "accept": + require_check_integrity(packet["checks"]) revisions_used = sum( entry["decision"]["disposition"] == "revise" for entry in store.branch_review_history(run_id) @@ -1753,7 +1756,10 @@ def handoff_complete( "headless review does not accept a host completion" ) if managed_review: - self._review_gate(store, run_id, handoff, snapshot, decision) + # The store still validates the submission hash/identity on a + # terminal replay. Do not apply new execution gates retroactively. + if run["state"] not in TERMINAL_STATES: + self._review_gate(store, run_id, handoff, snapshot, decision) submission = store.record_handoff_submission(run_id, decoded, decision) continuation = None if managed_review: diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py index 76928d5..c0496e8 100644 --- a/plugin/core/src/devsquad/validation.py +++ b/plugin/core/src/devsquad/validation.py @@ -54,7 +54,7 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) raise ContractError("checks must be a bounded array") check_ids = set() for check in value["checks"]: - _exact(check, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, "check") + _exact(check, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass", "output_paths"}, {"id", "argv", "cwd", "timeout_seconds", "required_to_pass"}, "check") if not isinstance(check["id"], str) or not check["id"]: raise ContractError("check id must be non-empty") if check["id"] in check_ids: raise ContractError("check ids must be unique") check_ids.add(check["id"]) @@ -62,6 +62,15 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) _relative(check["cwd"], "check cwd") if not isinstance(check["timeout_seconds"], int) or isinstance(check["timeout_seconds"], bool) or check["timeout_seconds"] <= 0: raise ContractError("check timeout must be positive") if type(check["required_to_pass"]) is not bool: raise ContractError("required_to_pass must be boolean") + outputs = check.get("output_paths", []) + if not isinstance(outputs, list) or len(outputs) > 32: + raise ContractError("check output_paths must be a bounded array") + for output in outputs: + _relative(output, "check output path") + if Path(output).as_posix() != output or output == "." or ".git" in Path(output).parts: + raise ContractError("check output path must be canonical and exclude Git metadata/root") + if len(set(outputs)) != len(outputs): + raise ContractError("check output paths must be unique") scope = value["scope"]; _exact(scope, {"read_paths", "write_paths"}, {"read_paths", "write_paths"}, "scope") if not isinstance(scope["read_paths"], list) or not isinstance(scope["write_paths"], list) or not all(isinstance(p, str) and p for p in scope["read_paths"] + scope["write_paths"]): raise ContractError("scope paths must be non-empty string arrays") if len(scope["read_paths"]) + len(scope["write_paths"]) > MAX_SCOPE_PATHS: raise ContractError("scope paths exceed their bound") diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index 3d30efd..a1fe301 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -20,7 +20,7 @@ MAX_TEXT_CHARS = 20_000 MAX_PREVIEW_CHARS = 1024 FINDING_SEVERITIES = {"critical", "high", "medium", "low"} -CHECK_STATUSES = {"passed", "failed", "timed_out", "launch_failed"} +CHECK_STATUSES = {"passed", "failed", "timed_out", "launch_failed", "invalidated", "not_run"} REVIEW_MODES = {"standard", "adversarial"} _SHA256 = re.compile(r"[0-9a-f]{64}\Z") _COMMIT_OID = re.compile(r"[0-9a-f]{40}\Z") @@ -301,12 +301,16 @@ def validate_check_results( candidate_sha256, _, target_oid = _workspace_identity(workspace) normalized = [] for configured, supplied in zip(task["checks"], values): - result = _exact(supplied, { + version = supplied.get("schema_version") if isinstance(supplied, dict) else None + fields = { "schema_version", "candidate_sha256", "target_oid", "id", "argv", "cwd", "required_to_pass", "status", "returncode", "error_code", "duration_ms", "stdout", "stderr", - }, "check result") - if result["schema_version"] != 1 or type(result["schema_version"]) is not int: + } + if version == 2: + fields.update({"integrity", "output_paths"}) + result = _exact(supplied, fields, "check result") + if type(version) is not int or version not in {1, 2}: raise ContractError("check result schema_version is invalid") if _sha256(result["candidate_sha256"], "check candidate_sha256") != candidate_sha256: raise ContractError("check result targets a different candidate hash") @@ -315,14 +319,63 @@ def validate_check_results( for field in ("id", "argv", "cwd", "required_to_pass"): if result[field] != configured[field]: raise ContractError(f"check result changes declared field: {field}") - if result["status"] not in CHECK_STATUSES: + if not isinstance(result["status"], str) or result["status"] not in CHECK_STATUSES: raise ContractError("check result status is invalid") if type(result["duration_ms"]) is not int or result["duration_ms"] < 0: raise ContractError("check duration_ms must be a non-negative integer") status, returncode, error_code = ( result["status"], result["returncode"], result["error_code"] ) - if status == "passed" and (returncode != 0 or error_code is not None): + if version == 2: + if result["output_paths"] != configured.get("output_paths", []): + raise ContractError("check result changes declared field: output_paths") + integrity = _exact(result["integrity"], { + "status", "reasons", "before_state_sha256", "after_state_sha256", + "changes", "changes_truncated", + }, "check integrity") + expected = {"invalidated": "violated", "not_run": "not_run"}.get(status, "verified") + if integrity["status"] != expected: + raise ContractError("check integrity contradicts its status") + reasons = integrity["reasons"] + if (not isinstance(reasons, list) or len(reasons) > 20 + or any(not isinstance(reason, str) or not reason or len(reason) > 200 + for reason in reasons) + or (expected == "verified") != (reasons == [])): + raise ContractError("check integrity reasons are inconsistent") + for field in ("before_state_sha256", "after_state_sha256"): + if status == "not_run": + if integrity[field] is not None: + raise ContractError("skipped check cannot claim an integrity observation") + else: + _sha256(integrity[field], f"check integrity {field}") + changes = integrity["changes"] + if (not isinstance(changes, list) or len(changes) > 100 + or type(integrity["changes_truncated"]) is not bool): + raise ContractError("check integrity changes must be bounded") + if status != "invalidated" and (changes or integrity["changes_truncated"]): + raise ContractError("clean/skipped check cannot report input changes") + if expected == "verified" and integrity["before_state_sha256"] != integrity["after_state_sha256"]: + raise ContractError("verified check has differing input fingerprints") + for change in changes: + _exact(change, {"workspace", "path", "before", "after"}, "check integrity change") + if not isinstance(change["workspace"], str) or change["workspace"] not in {"check", "review"}: + raise ContractError("check integrity change workspace is invalid") + path = _text(change["path"], "check integrity path", maximum=4096) + if PurePosixPath(path).is_absolute() or ".." in PurePosixPath(path).parts: + raise ContractError("check integrity path escapes workspace") + for field in ("before", "after"): + if change[field] is not None and change[field] != "undeclared": + _sha256(change[field], f"check change {field}") + if change["before"] == change["after"]: + raise ContractError("check integrity change has identical evidence") + elif status in {"invalidated", "not_run"}: + raise ContractError("check integrity requires schema version 2") + if status in {"invalidated", "not_run"} and ( + error_code != "CLI_ERROR" + or (returncode is not None and type(returncode) is not int) + or (status == "not_run" and returncode is not None)): + raise ContractError("invalidated/skipped check result is inconsistent") + if status == "passed" and (type(returncode) is not int or returncode != 0 or error_code is not None): raise ContractError("passing check result is inconsistent") if status == "failed" and ( type(returncode) is not int or returncode == 0 or error_code is not None @@ -376,6 +429,10 @@ def evaluate_branch_review( "evidence": evidence, }) candidate_sha256, base_oid, target_oid = _workspace_identity(workspace) + integrity_failures = [ + result["id"] for result in normalized_checks + if result.get("integrity", {}).get("status") in {"violated", "not_run"} + ] return { "schema_version": 1, "candidate_sha256": candidate_sha256, @@ -385,10 +442,10 @@ def evaluate_branch_review( "required_checks_passed": not required_failures, "required_failures": required_failures, "report_only_failures": report_only_failures, - "accept_allowed": not required_failures, + "accept_allowed": not required_failures and not integrity_failures, "accept_blockers": [ f"required_check_failed:{check_id}" for check_id in required_failures - ], + ] + [f"candidate_integrity_failed:{check_id}" for check_id in integrity_failures], "criteria": criteria, } @@ -1027,12 +1084,23 @@ def decode_branch_review_evidence( payload: bytes | str, snapshot: dict[str, Any], ) -> dict[str, Any]: - return validate_branch_review_evidence( + evidence = validate_branch_review_evidence( _strict_json_object( payload, "branch review evidence", maximum=MAX_EVIDENCE_BYTES, ), snapshot, ) + require_check_integrity(evidence["checks"]) + return evidence + + +def require_check_integrity(checks: list[dict[str, Any]]) -> None: + """Keep v1 receipts readable, but never use them for a new acceptance.""" + if any(check.get("schema_version") != 2 for check in checks): + raise ContractError( + "new acceptance requires candidate integrity verification; " + "start a new run to refresh legacy check evidence" + ) def validate_branch_review_handoff( diff --git a/plugin/core/src/devsquad/workspaces.py b/plugin/core/src/devsquad/workspaces.py index 9bc4051..a5ebed4 100644 --- a/plugin/core/src/devsquad/workspaces.py +++ b/plugin/core/src/devsquad/workspaces.py @@ -6,8 +6,9 @@ import fcntl import os from pathlib import Path, PurePosixPath +import stat import subprocess -from typing import Iterable +from typing import Any, Iterable from .contracts import ContractError from .store import canonical_json, git_common_dir @@ -227,6 +228,69 @@ def reset_check_workspace( _validate_workspace(review, checks, target_oid, scope_paths) +def candidate_input_state( + workspace: Path, target_oid: str, output_paths: Iterable[str] = (), +) -> dict[str, Any]: + """Fingerprint tracked inputs without trusting index stat/skip-worktree hints. + + Only explicitly approved untracked outputs are excluded. Inventory comes from the + frozen commit, so removing a file from the index cannot hide its mutation. + Hash actual checked-out bytes (including clean/smudge transformations), not + Git's possibly cached diff. This is compared across each check boundary. + """ + root = workspace.resolve(strict=True) + top = Path(_git(root, "rev-parse", "--show-toplevel").decode().strip()).resolve() + if top != root: + raise ContractError("candidate workspace top level changed") + files = [] + for record in _git(root, "ls-tree", "-r", "-z", target_oid).split(b"\0"): + if not record: + continue + metadata, raw_name = record.split(b"\t", 1) + mode, kind, _ = metadata.split() + name = _decode_paths(raw_name + b"\0", "candidate inventory")[0] + if kind != b"blob" or mode not in {b"100644", b"100755", b"120000"}: + raise ContractError("candidate integrity does not support Git submodules") + path = root / name + # A symlink replacing a tracked parent directory must not redirect reads. + parent = path.parent.resolve(strict=True) + if parent != path.parent: + raise ContractError("candidate tracked parent became a symlink") + try: + info = path.lstat() + except FileNotFoundError: + files.append([name, "missing"]) + continue + if stat.S_ISLNK(info.st_mode): + files.append([name, "symlink", os.readlink(path)]) + elif stat.S_ISREG(info.st_mode): + digest = hashlib.sha256() + with path.open("rb") as stream: + for block in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(block) + files.append([name, "file", bool(info.st_mode & 0o111), digest.hexdigest()]) + else: + files.append([name, "unsupported"]) + outputs = tuple(output_paths) + unapproved = sorted( + name for name in _decode_paths(_git(root, "ls-files", "--others", "-z"), "untracked inputs") + if not any(_intersects(name, output) for output in outputs) + or not (root / name).resolve().is_relative_to(root) + ) + return { + "worktree": str(root) + "\n" + str(git_common_dir(root)), + "head": resolve_commit(root, "HEAD"), + "branch": _git(root, "rev-parse", "--abbrev-ref", "HEAD").decode().strip(), + "index": hashlib.sha256(_git(root, "ls-files", "--stage", "-v", "-z")).hexdigest(), + "tracked_inputs": hashlib.sha256(canonical_json(files).encode()).hexdigest(), + "undeclared_inputs": hashlib.sha256(canonical_json(unapproved).encode()).hexdigest(), + "paths": { + entry[0]: hashlib.sha256(canonical_json(entry[1:]).encode()).hexdigest() + for entry in files + } | {name: "undeclared" for name in unapproved}, + } + + def _prepare_detached_workspace( source_repo: Path, runtime: Path, diff --git a/test/core/test_check_integrity_runtime.py b/test/core/test_check_integrity_runtime.py new file mode 100644 index 0000000..448632a --- /dev/null +++ b/test/core/test_check_integrity_runtime.py @@ -0,0 +1,331 @@ +"""Candidate-integrity gates exercised through durable public service runs.""" + +from __future__ import annotations + +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import time +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.store import ConflictError, Store, request_hash +import test_delivery_workflow as delivery_fixtures + + +class PublicCheckIntegrityTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-check-integrity-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + subprocess.run(["git", "init", "-q", str(self.repo)], check=True) + self.git("config", "user.email", "fixture@example.invalid") + self.git("config", "user.name", "Fixture") + for name in ("src", "tests", "devsquad"): + (self.repo / name).mkdir() + (self.repo / "src/app.py").write_text("VALUE = 'base'\n") + (self.repo / "tests/test_app.py").write_text("# fixture\n") + profiles, policy = delivery_fixtures.DeliveryWorkspaceTest.delivery_routing_documents() + policy["task_classes"]["fixture-review-small"] = "proven" + for name, document in (("profiles", profiles), ("policy", policy)): + (self.repo / "devsquad" / f"{name}.json").write_text(json.dumps(document)) + self.git("add", ".") + self.git("commit", "-qm", "baseline") + self.baseline = self.git("rev-parse", "HEAD").strip() + (self.repo / "src/app.py").write_text("VALUE = 'wrong'\n") + self.git("add", "src/app.py") + self.git("commit", "-qm", "candidate needing a fix") + self.target = self.git("rev-parse", "HEAD").strip() + self.original = { + "head": self.target, + "index": self.git("write-tree"), + "status": self.git("status", "--porcelain"), + "refs": self.git("show-ref"), + "source": (self.repo / "src/app.py").read_bytes(), + } + self.service = Service(self.runtime) + self.run_ids = [] + self.addCleanup(self.cancel_unfinished_runs) + + def git(self, *arguments): + return subprocess.run( + ["git", "-C", str(self.repo), *arguments], check=True, + text=True, capture_output=True, + ).stdout + + def cancel_unfinished_runs(self): + for run_id in self.run_ids: + if self.service.status(run_id)["state"] not in {"succeeded", "failed", "cancelled"}: + self.service.cancel(run_id) + + def task(self, workflow, mode, checks): + task = json.loads( + (ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text() + ) + task["project"] = { + "repo_path": str(self.repo), "base_ref": self.baseline, + "target_ref": self.target if workflow == "branch-review" else self.baseline, + } + task["workflow"] = workflow + task["lead"] = {"mode": mode} + task["checks"] = checks + task["budget"] = { + "wall_seconds": 60, "max_worker_invocations": 3, + "max_revisions": 0, "max_fallbacks_per_step": 0, + } + if workflow == "issue-delivery": + task["task_class"] = "fixture-delivery-small" + task["scope"]["write_paths"] = ["src/app.py"] + return task + + @staticmethod + def check(check_id, script, *arguments, required=False, output_paths=None): + result = { + "id": check_id, + "argv": [sys.executable, "-c", script, *arguments], + "cwd": ".", "timeout_seconds": 10, "required_to_pass": required, + } + if output_paths is not None: + result["output_paths"] = output_paths + return result + + def wait_state(self, run_id, states): + deadline = time.monotonic() + 20 + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] in states: + return status + time.sleep(0.05) + log = self.runtime / "private-logs" / f"{run_id}.supervisor.log" + self.fail( + f"run did not reach {states}: {self.service.status(run_id)}\n" + f"{log.read_text() if log.exists() else 'no supervisor log'}" + ) + + def start(self, workflow, mode, checks): + options = { + "_internal_review_fixture": { + "verdict": "clean", "summary": "Fixture review of the exact candidate.", + "findings": [], + }, + } + if workflow == "issue-delivery": + options["_internal_implementation_fixture"] = { + "writes": [{"path": "src/app.py", "content": "VALUE = 'wrong'\n"}], + "delay_seconds": 0, + } + if mode == "headless": + options["_internal_lead_fixture"] = { + "disposition": "accept", "reason": "Attempt to accept the saved evidence.", + } + started = self.service.start( + self.task(workflow, mode, checks), f"integrity-{workflow}-{mode}", **options, + ) + run_id = started["run_id"] + self.run_ids.append(run_id) + self.assertTrue(started["created"]) + if workflow == "issue-delivery": + deadline = time.monotonic() + 20 + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] == "queued" and status.get("next_action") == "resume_candidate_review": + break + self.assertNotIn(status["state"], {"failed", "cancelled"}, status) + time.sleep(0.05) + else: + self.fail(f"delivery candidate was not frozen: {self.service.status(run_id)}") + self.assertTrue(self.service.resume(run_id)["launched"]) + return run_id + + @staticmethod + def decision(packet, submission_id, disposition): + body = { + "schema_version": 1, "submission_id": submission_id, + "disposition": disposition, "reason": "Exercise the candidate integrity gate.", + "evidence_refs": [ + {"artifact_id": item["artifact_id"], "sha256": item["sha256"]} + for item in packet["artifacts"] + ], + } + return {**body, "submission_hash": request_hash(body)} + + def receipt(self, run_id): + artifact = next( + item for item in self.service.result(run_id)["artifacts"] + if item["name"] == "receipt.json" + ) + return json.loads(Path(artifact["path"]).read_text()) + + def snapshot(self, run_id): + store = Store(self.service.database, self.service.artifacts) + try: + return json.loads(store.run(run_id)["mutable_snapshot"]) + finally: + store.close() + + def assert_original_unchanged(self): + self.assertEqual(self.git("rev-parse", "HEAD").strip(), self.original["head"]) + self.assertEqual(self.git("write-tree"), self.original["index"]) + self.assertEqual(self.git("status", "--porcelain"), self.original["status"]) + self.assertEqual(self.git("show-ref"), self.original["refs"]) + self.assertEqual((self.repo / "src/app.py").read_bytes(), self.original["source"]) + + def exercise_mutation(self, workflow, mode, *, add_source=False): + second_marker = self.root / "second-check-ran" + mutated_path = "src/helper.py" if add_source else "src/app.py" + expected_output = { + "before": None if add_source else "VALUE = 'wrong'\n", + "after": "VALUE = 'fixed'\n", + } + checks = [ + self.check( + "mutating-report-only-check", + "from pathlib import Path; import json,sys; p=Path(sys.argv[1]); " + "before=p.read_text() if p.exists() else None; " + "p.write_text(\"VALUE = 'fixed'\\n\"); " + "print(json.dumps({'before':before,'after':p.read_text()}))", + mutated_path, + ), + self.check( + "must-not-use-contaminated-source", + "from pathlib import Path; import sys; " + "Path(sys.argv[1]).write_text(Path(sys.argv[2]).read_text())", + str(second_marker), mutated_path, + ), + ] + run_id = self.start(workflow, mode, checks) + status = self.wait_state(run_id, {"awaiting_host", "succeeded", "failed"}) + if mode == "host": + self.assertEqual(status["state"], "awaiting_host", status) + claimed = self.service.handoff_claim(run_id, status["version"], "integrity-host") + evidence = claimed["handoff"]["packet"] + else: + evidence = self.receipt(run_id) + self.assertEqual(status["state"], "failed", "headless lead accepted changed source") + + # A successful shell exit is not a successful check of the frozen source. + self.assertIsInstance(evidence["evaluation"], dict, evidence) + self.assertFalse(evidence["evaluation"]["accept_allowed"]) + self.assertFalse(second_marker.exists(), "later check ran on mutated candidate source") + self.assertEqual(evidence["checks"][0]["status"], "invalidated") + self.assertEqual(evidence["checks"][0]["returncode"], 0) + self.assertEqual(evidence["checks"][0]["integrity"]["status"], "violated") + integrity = evidence["checks"][0]["integrity"] + self.assertNotEqual(integrity["before_state_sha256"], integrity["after_state_sha256"]) + self.assertEqual(integrity["changes"][0]["path"], mutated_path) + self.assertNotEqual(integrity["changes"][0]["before"], integrity["changes"][0]["after"]) + self.assertEqual(evidence["checks"][1]["status"], "not_run") + self.assertIsNone(evidence["checks"][1]["returncode"]) + self.assertEqual(evidence["checks"][1]["integrity"]["status"], "not_run") + output = json.loads(evidence["checks"][0]["stdout"]["preview"]) + self.assertEqual(output, expected_output) + snapshot = self.snapshot(run_id) + candidate_oid = snapshot["workspace"]["target_oid"] + self.assertEqual(self.git("show", f"{candidate_oid}:src/app.py"), "VALUE = 'wrong'\n") + if add_source: + self.assertEqual(self.git("ls-tree", candidate_oid, "--", mutated_path), "") + self.assertEqual( + (Path(snapshot["check_workspace"]["path"]) / mutated_path).read_text(), + "VALUE = 'fixed'\n", + ) + + # A new service instance must replay the invalid evidence, not rerun it + # against the altered tree or lose the gate when recovering a handoff. + self.service = Service(self.runtime) + with self.assertRaises(ConflictError): + self.service.resume(run_id) + if mode == "host": + with self.assertRaisesRegex(ContractError, "blocked by required evidence"): + self.service.handoff_complete( + run_id, claimed["claim"], self.decision(evidence, "invalid-accept", "accept"), + ) + terminal = self.service.handoff_complete( + run_id, claimed["claim"], self.decision(evidence, "reject-mutation", "reject"), + ) + self.assertEqual(terminal["state"], "failed") + receipt = self.receipt(run_id) + self.assertFalse(receipt["evaluation"]["accept_allowed"]) + self.assertEqual(receipt["checks"], evidence["checks"]) + with self.assertRaises(ConflictError): + self.service.resume(run_id) + self.assertEqual(self.service.status(run_id)["state"], "failed") + self.assertFalse(second_marker.exists()) + self.assert_original_unchanged() + + def test_branch_review_host_rejects_source_mutating_report_only_check(self): + self.exercise_mutation("branch-review", "host") + + def test_branch_review_headless_rejects_source_mutating_report_only_check(self): + self.exercise_mutation("branch-review", "headless") + + def test_delivery_host_rejects_source_mutating_report_only_check(self): + self.exercise_mutation("issue-delivery", "host") + + def test_delivery_headless_rejects_source_mutating_report_only_check(self): + self.exercise_mutation("issue-delivery", "headless") + + def test_branch_review_host_rejects_new_untracked_source_input(self): + self.exercise_mutation("branch-review", "host", add_source=True) + + def test_delivery_host_rejects_new_untracked_source_input(self): + self.exercise_mutation("issue-delivery", "host", add_source=True) + + def exercise_build_output(self, workflow): + checks = [ + self.check( + "ordinary-build-output", + "from pathlib import Path; Path('tests/build-output.bin').write_bytes(b'build')", + required=True, + output_paths=["tests/build-output.bin"], + ), + self.check( + "clean-source-and-build-output", + "from pathlib import Path; " + "assert Path('src/app.py').read_text() == \"VALUE = 'wrong'\\n\"; " + "assert Path('tests/build-output.bin').read_bytes() == b'build'", + required=True, + ), + ] + run_id = self.start(workflow, "headless", checks) + self.assertEqual(self.wait_state(run_id, {"succeeded", "failed"})["state"], "succeeded") + receipt = self.receipt(run_id) + self.assertTrue(receipt["evaluation"]["accept_allowed"]) + self.assertEqual([check["status"] for check in receipt["checks"]], ["passed", "passed"]) + self.assert_original_unchanged() + + def test_branch_review_permits_untracked_build_output_and_clean_checks(self): + self.exercise_build_output("branch-review") + + def test_delivery_permits_untracked_build_output_and_clean_checks(self): + self.exercise_build_output("issue-delivery") + + def test_terminal_completion_replay_preserves_historical_acceptance(self): + run_id = self.start("branch-review", "host", [self.check("clean", "print('ok')")]) + status = self.wait_state(run_id, {"awaiting_host", "failed"}) + self.assertEqual(status["state"], "awaiting_host") + claimed = self.service.handoff_claim(run_id, status["version"], "replay-host") + decision = self.decision(claimed["handoff"]["packet"], "accept-once", "accept") + first = self.service.handoff_complete(run_id, claimed["claim"], decision) + self.assertEqual(first["state"], "succeeded") + receipt = self.receipt(run_id) + with patch("devsquad.service.require_check_integrity", side_effect=ContractError("legacy check")) as gate: + replay = self.service.handoff_complete(run_id, claimed["claim"], decision) + gate.assert_not_called() + self.assertTrue(replay["replayed"]) + self.assertEqual(self.receipt(run_id), receipt) + conflicting = self.decision(claimed["handoff"]["packet"], "accept-once", "reject") + with self.assertRaises(ConflictError): + self.service.handoff_complete(run_id, claimed["claim"], conflicting) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index 2d06f69..ce25afe 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -436,6 +436,7 @@ def test_revision_rechecks_clean_workspace_and_preserves_both_attempts(self): "cwd": ".", "timeout_seconds": 10, "required_to_pass": True, + "output_paths": ["generated.tmp"], }] run_id, first_wait = self.start_waiting("one-revision") first_claim = self.service.handoff_claim( diff --git a/test/core/test_review_workflow.py b/test/core/test_review_workflow.py index d916a9b..0697bff 100644 --- a/test/core/test_review_workflow.py +++ b/test/core/test_review_workflow.py @@ -13,9 +13,11 @@ apply_lead_disposition, build_review_prompt, decode_review_document, + decode_branch_review_evidence, evaluate_branch_review, make_branch_review_evidence, review_output_schema, + require_check_integrity, validate_branch_review_evidence, validate_check_results, validate_review_document, @@ -211,6 +213,35 @@ def test_check_status_and_bounded_stream_metadata_are_consistent(self): with self.assertRaisesRegex(ContractError, "truncation metadata"): validate_check_results([invalid], self.task, self.workspace) + def test_integrity_is_non_overridable_and_strictly_bound_to_frozen_outputs(self): + check = self.check_result("invalidated", returncode=0, error_code="CLI_ERROR") + check.update(schema_version=2, output_paths=[], integrity={ + "status": "violated", "reasons": ["check:tracked_inputs_changed"], + "before_state_sha256": "1" * 64, "after_state_sha256": "2" * 64, + "changes": [], "changes_truncated": False, + }) + evaluation = evaluate_branch_review(self.task, self.workspace, self.review, [check]) + self.assertTrue(evaluation["required_checks_passed"]) + self.assertFalse(evaluation["accept_allowed"]) + with self.assertRaises(ContractError): + apply_lead_disposition(evaluation, "accept", revisions_used=0, max_revisions=1) + for field, value in (("status", "passed"), ("output_paths", ["src"])): + tampered = copy.deepcopy(check) + tampered[field] = value + with self.assertRaises(ContractError): + validate_check_results([tampered], self.task, self.workspace) + check["integrity"]["reasons"] = [] + with self.assertRaisesRegex(ContractError, "reasons"): + validate_check_results([check], self.task, self.workspace) + + def test_legacy_checks_remain_readable_but_cannot_be_imported_or_accepted_anew(self): + evidence = make_branch_review_evidence(self.snapshot, self.review, [self.check]) + self.assertEqual(validate_branch_review_evidence(evidence, self.snapshot), evidence) + with self.assertRaisesRegex(ContractError, "integrity verification"): + decode_branch_review_evidence(json.dumps(evidence), self.snapshot) + with self.assertRaisesRegex(ContractError, "start a new run"): + require_check_integrity(evidence["checks"]) + def test_report_only_failure_is_visible_but_does_not_block_delivery(self): evaluation = evaluate_branch_review( self.task, self.workspace, self.review, [self.check], diff --git a/test/core/test_validation.py b/test/core/test_validation.py index 5a3091d..9cf7900 100644 --- a/test/core/test_validation.py +++ b/test/core/test_validation.py @@ -65,6 +65,17 @@ def test_task_rejects_adversarial_nested_types(self): value = copy.deepcopy(self.task); mutate(value) self.assert_contract_error(validate_task, value) + def test_check_outputs_are_explicit_bounded_relative_paths(self): + valid = copy.deepcopy(self.task) + valid["checks"][0]["output_paths"] = ["build", "tests/result.json"] + validate_task(valid) + for outputs in ("build", ["."], ["../escape"], ["/tmp/out"], + [".git"], ["build/../src"], ["./build"], ["build", "build"], + ["build/"] , [None], [f"out-{i}" for i in range(33)]): + invalid = copy.deepcopy(self.task) + invalid["checks"][0]["output_paths"] = outputs + self.assert_contract_error(validate_task, invalid) + def test_task_accepts_exact_embedded_routing_and_rejects_mixed_sources(self): profiles = { "schema_version": 1, diff --git a/test/core/test_workspaces.py b/test/core/test_workspaces.py index 766a901..af00666 100644 --- a/test/core/test_workspaces.py +++ b/test/core/test_workspaces.py @@ -1,4 +1,5 @@ import hashlib +import json from pathlib import Path import subprocess import sys @@ -9,6 +10,7 @@ sys.path.insert(0, str(ROOT / "plugin/core/src")) from devsquad.contracts import ContractError +from devsquad.review_worker import run_review_and_checks from devsquad.workspaces import ( assert_clean_inputs, committed_regular_file, @@ -69,6 +71,61 @@ def git(self, *args): stdout=subprocess.PIPE, ).stdout + def test_check_integrity_covers_tracked_modes_index_head_and_hidden_changes(self): + mutations = { + "delete": "p.unlink()", + "mode": "p.chmod(0o755)", + "staged": "p.write_text('changed\\n'); git('add','src/app.py')", + "index-only": "git('update-index','--chmod=+x','src/app.py')", + "hidden": "git('update-index','--assume-unchanged','src/app.py'); p.write_text('hidden\\n')", + "checkout": f"git('checkout','--detach','{self.base}')", + "attach": "git('checkout','-b','check-attached-branch')", + "symlink": "p.unlink(); p.symlink_to('../tests/test_app.py')", + "ignored-source": "Path('.gitignore').write_text('src/hidden.py\\n'); Path('src/hidden.py').write_text('hidden\\n')", + "review-tree": "(Path.cwd().parent/'review-worktree/src/app.py').write_text('changed review\\n')", + } + for name, mutation in mutations.items(): + with self.subTest(name=name): + workspace = self.prepare(run_id=name) + checks_workspace = prepare_check_workspace( + self.repo, self.runtime, "project-1", name, self.target, ("src", "tests"), + ) + task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + script = ( + "from pathlib import Path; import subprocess; p=Path('src/app.py'); " + "git=lambda *args: subprocess.run(['git',*args],check=True,capture_output=True); " + + mutation + "; print('mutation completed')" + ) + task["checks"] = [{ + "id": "mutation", "argv": [sys.executable, "-c", script], "cwd": ".", + "required_to_pass": False, "timeout_seconds": 10, + # Listing tracked source as an output never permits editing it. + "output_paths": ["src/app.py"], + }, { + "id": "later", "argv": [sys.executable, "-c", "print('must not run')"], + "cwd": ".", "required_to_pass": False, "timeout_seconds": 10, + }] + review = { + "schema_version": 1, "candidate_sha256": workspace["candidate_sha256"], + "base_oid": self.base, "target_oid": self.target, + "review_mode": "standard", "verdict": "clean", "summary": "Fixture", "findings": [], + } + selected = { + "profile_id": "fixture-reviewer", "profile_sha256": "1" * 64, + "profile": {"harness": "fixture"}, + "reference": {"kind": "profile", "id": "fixture-reviewer"}, "binding": None, + } + evidence = run_review_and_checks({ + "task": task, "workspace": workspace, "check_workspace": checks_workspace, + "routing": {"roles": {"reviewer": {"selected": selected}}}, + }, review) + self.assertEqual(evidence["checks"][0]["returncode"], 0) + self.assertEqual(evidence["checks"][0]["status"], "invalidated") + self.assertEqual(evidence["checks"][1]["status"], "not_run") + self.assertEqual(evidence["checks"][1]["stdout"]["total_bytes"], 0) + self.assertFalse(evidence["evaluation"]["accept_allowed"]) + self.assertEqual((self.repo / "src/app.py").read_text(), "VALUE = 'candidate'\n") + def test_workspace_is_detached_frozen_idempotent_and_checkout_preserving(self): (self.repo / "notes.txt").write_text("dirty but outside declared inputs\n") before_head = self.git("rev-parse", "HEAD") From 36129428eb75936962a6e2c11957696c948ccb7c Mon Sep 17 00:00:00 2001 From: Dikshant Date: Tue, 29 Sep 2026 08:28:11 -0700 Subject: [PATCH 133/197] WIP checkpoint: Sol execution plan and R2 failing regression handoff (2026-09-29 08:28) --- docs/plans/engineering-team/M5-STATUS.md | 7 + docs/plans/engineering-team/RESUME.md | 22 +- .../engineering-team/SOL-REVIEW-FOLLOWUP.md | 89 ++++- docs/plans/engineering-team/backlog.json | 2 +- .../R2-identity-red-baseline-2026-09-29.json | 39 +++ test/core/test_claude_identity.py | 311 ++++++++++++++++++ 6 files changed, 462 insertions(+), 8 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R2-identity-red-baseline-2026-09-29.json create mode 100644 test/core/test_claude_identity.py diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index 271618b..6aea58b 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -7,6 +7,13 @@ and verified offline; execute R2/R6 in [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The genuine Claude-to-Codex live gate additionally remains blocked on normal Claude login. +R2 now has a preserved red baseline: 16 fake-CLI worker tests ran with 39 +assertion/subtest failures and 4 errors against unchanged production source +at `6848f11`. See [baseline evidence](evidence/R2-identity-red-baseline-2026-09-29.json). +Parser repair, strict imported evidence, public independent-review acceptance +and durable failure/replay regressions remain unfinished. This checkpoint +does not establish a green core suite or a corrected Claude live receipt. + | Requirement | Planned evidence | Status | |---|---|---| | Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | argv conformance verified at `d96e9e4`; observed identity reopened under R2 | diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 18c798a..c9f17b1 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,22 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Review correction and next action +Latest handoff is a **plan plus preserved failing R2 regressions**, not an R2 +repair. Production remains at R1 commit `6848f11`. +`test/core/test_claude_identity.py` now contains 16 fake-CLI tests; the focused +baseline ran in 3.162 seconds with 39 assertion/subtest failures and 4 errors +(non-object JSON raises `AttributeError`). No live provider call or new +production change occurred. The current test tree is knowingly red; the +330-test green result below predates these new regressions. See +[R2 baseline evidence](evidence/R2-identity-red-baseline-2026-09-29.json). +This handoff's Bash gate passed all 227 assertions across 11 files with +process inspection permitted; the first sandbox-stalled run was stopped and +is not a pass. Generated-reference, JSON and whitespace checks passed. No +complete core rerun was needed for this plan-only checkpoint; R2 remains red. +Sol's next slice is R2a parser/worker, then R2b strict import/actual-model +independence and R2c failed-receipt/replay coverage, as detailed in the +[updated execution plan](SOL-REVIEW-FOLLOWUP.md#ready-to-execute-handoff-for-sol). + R1's source repair is verified after `399d93d`: the public regression first reproduced four unsafe mutation paths (review/delivery × host/headless). The repair adds check-boundary fingerprints, explicit permitted output paths, @@ -314,8 +330,10 @@ provider paths must not be advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Execute R2's offline Claude model-identity regressions and strict import gate, - then R3 and R4–R6 in dependency order. Preserve the verified R1 repair, +2. Implement R2 against the saved 16-test failing Claude-identity baseline; + add strict import, actual-model independence and durable failure/replay + public regressions. Then execute R3 and R4–R6 in dependency order. Preserve + the verified R1 repair, explicit check `output_paths` contract and historical receipts. Do not rewrite the architecture or reset completed work. 3. Keep Claude and Grok live probes paused until normal login is restored. diff --git a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md index d797410..a9b5160 100644 --- a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md +++ b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md @@ -26,6 +26,33 @@ coverage and 330-test core gate (two optional-SDK skips). See [R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json). Start at **R2** after checking current Git state. R2–R8 and installed refresh remain open. +### Ready-to-execute handoff for Sol + +Production baseline is now `6848f1156bee10e1ef2617402dfc7eaebdc58076` (R1). +R2 has a saved **red test checkpoint**, not an implemented repair: +`test/core/test_claude_identity.py` contains 16 offline fake-CLI tests. The +latest focused run completed in 3.162 seconds with **39 assertion/subtest +failures and 4 errors**. The errors expose non-object JSON reaching an unsafe +`.get()` call. See [R2 baseline evidence](evidence/R2-identity-red-baseline-2026-09-29.json). +No production source changed for this planning handoff. The prior 330-test +green gate belongs to R1; the current tree includes known failing R2 tests. + +| Next slice | Deliverable | Gate before claiming completion | +|---|---|---| +| R2a | Native identity parser and worker evidence | Existing 16 tests green; explicit provenance and unknown settings | +| R2b | Strict import plus actual-model independence | Tampered evidence and same effective model rejected through public host/headless paths | +| R2c | Durable identity failures and replay compatibility | Invalid identity retained in failed receipts; historical reads work without authorizing new acceptance | +| R3 | Independent trial evidence | No reused observation can inflate evaluation or held-out sample size | +| R4 → R5 → R6 | Routing, learning and normal terminal integration | Public-run evidence through the full chain, then fresh-install usability proof | +| R7 / R8 | Council and installed/live closure | Each separately required acceptance gate; external blockers remain explicit | + +Use the same spec-based loop for every slice: name the contract and public +entry point; reproduce the missing behavior; implement the smallest coherent +change; run focused tests; review the patch; record evidence and the exact +next action; checkpoint. Run the full integration gates at package boundaries, +not after every small edit. One integration owner controls shared files; use +bounded leaf reviews only when they can run alongside useful local work. + ## Feedback on what has been built Preserve the substantial working implementation: the durable runner and @@ -136,6 +163,55 @@ model, absent identity, multiple models, unavailable effort and tampered evidence all have explicit tested results. The genuine two-harness live gate remains open until the corrected adapter obtains a real receipt. +#### R2 implementation feedback and missing tests + +- **R2a — parser/worker:** define a shared strict identity contract rather than + letting the worker and importer invent separate interpretations. Require a + valid success envelope, bounded nonblank session identity and well-formed + native usage. Reject non-object, duplicate-key and non-finite JSON without + leaking an `AttributeError`. Keep the frozen requested profile unchanged. + A sole concrete `modelUsage` key may establish the reported model; preserve + all entries and reject ambiguous writer attribution. Tested family aliases + (`sonnet`, `opus`, `haiku`) may resolve only to a reported concrete model of + that family, without a hardcoded current version. Pricing `canonicalModel` + must not replace native usage identity; a differing pricing-only value is + allowed and is not a serving-identity conflict. An optional top-level + `model` is not identity authority either, but a contradictory serving-model + claim fails closed. Keep effective effort/backing revision + null when unreported, and make the scope of verification explicit. The new + test file covers this layer only; update older success fixtures to include + valid native evidence instead of weakening the parser. +- **R2b — import/acceptance:** validate the native evidence against the frozen + adapter, selected profile (including a legitimate fallback), session and + claimed observation. A dictionary or `verification="verified"` label is not + proof. Add regressions for forged model/effort/session/provenance and missing + native evidence. Enforce independence using verified reported identities, + not just requested aliases or family labels. Exercise same effective model + behind different profile names, unknown/ambiguous identity and a valid + different-model case through both host and headless delivery. Candidate + finalization/import and later disposition must not bypass these gates. +- **R2c — failures/recovery:** persist bounded, redacted identity diagnostics, + valid reported model/usage data and artifact references for rejected attempts + in the durable receipt; exception text alone is insufficient. Test failure, + fallback, cancellation and recovery/replay with no duplicate writer or lost + attempt. Historical terminal receipts remain readable and exact terminal + replays remain idempotent, but old unverified evidence cannot authorize a + new acceptance. Version evidence/contracts explicitly if needed. Preserve + R1's non-overridable check-integrity gates throughout. + +Reproduce the saved baseline with: + +```bash +PYTHONDONTWRITEBYTECODE=1 PYTHONPATH=plugin/core/src \ + python3 -m unittest discover -s test/core -p test_claude_identity.py -v +``` + +This command is expected to fail until R2 is implemented. Do not skip or mark +the regressions expected-failure merely to recover a green suite. Before R2 +offline closure, run its added public tests, complete core discovery, the Bash +suite and generated-reference check, then obtain one bounded independent +patch review. Only after that consider the separately blocked live receipt. + ### R3 — Make experiment and held-out evidence independent **Files:** `learning.py`, experiment/qualification paths in `store.py`, and @@ -311,8 +387,11 @@ permission to claim routing improvement. ## First action for Sol -R1's source repair is verified; continue at R2 with failing native-identity -and import-tampering regressions. Preserve R1's check-output contract and -mutation evidence. Continue R2–R6 without waiting on the Claude login or Jev key. Retain R7 -and R8 in the full scope and continue their independent work as dependencies -become ready. +Read the recovery files, verify current Git state, and start with the saved +16-test R2 red baseline above. Implement R2a, then add the missing R2b/R2c +public regressions and repairs. Preserve R1's check-output contract and +mutation evidence. Continue R2–R6 without waiting on the Claude login or Jev +key. Retain R7 and R8 in the full scope and continue their independent work +as dependencies become ready. Report each slice as red baseline, verified +offline, installed proof, or externally blocked; do not collapse those states +into a single completion claim. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 198284d..375af96 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -23,7 +23,7 @@ }, "review_work_packages": [ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, - {"id": "R2", "title": "Observed Claude execution identity", "status": "pending", "milestones": ["M5"], "depends_on": [], "items": ["F2"]}, + {"id": "R2", "title": "Observed Claude execution identity", "status": "in_progress", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-identity-red-baseline-2026-09-29.json", "checkpoint": "16 offline worker regressions saved; focused run reports 39 assertion/subtest failures and 4 errors against unchanged R1 production source. No R2 repair or green gate yet. Next: parser/worker, strict import/actual-model independence, then durable failure and replay coverage."}, {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "pending", "milestones": ["M6"], "depends_on": [], "items": ["F3"]}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, diff --git a/docs/plans/engineering-team/evidence/R2-identity-red-baseline-2026-09-29.json b/docs/plans/engineering-team/evidence/R2-identity-red-baseline-2026-09-29.json new file mode 100644 index 0000000..b8d9f37 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R2-identity-red-baseline-2026-09-29.json @@ -0,0 +1,39 @@ +{ + "schema_version": 1, + "work_package": "R2", + "recorded_on": "2026-09-29", + "status": "red_baseline_only", + "production_revision": "6848f1156bee10e1ef2617402dfc7eaebdc58076", + "production_changed": false, + "provider_calls": 0, + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -p test_claude_identity.py -v", + "result": { + "exit_code": 1, + "test_methods_run": 16, + "assertion_or_subtest_failures": 39, + "errors": 4, + "elapsed_seconds": 3.162, + "error_cause": "Non-object JSON reaches provider_document.get and raises AttributeError instead of ContractError" + }, + "sha256": { + "test/core/test_claude_identity.py": "c18f2f91ff547f2e3c90902bb3516550e20c39e8c9a01f7c8b8ec3ad8f2a2d99", + "plugin/core/src/devsquad/claude_delivery_worker.py": "c89bca4fa04e01541d1877c8ee8511df724e46dc00b6e172add32d03094de18c", + "plugin/core/src/devsquad/workflows.py": "449237aa230e6a778be42f21f931106beeb15326f1a4fd5054b99fce5e47b7dc" + }, + "scope": "Direct worker conformance using only a temporary fake CLI; not public delivery, installed runtime, or live-provider proof", + "handoff_checks": { + "bash_test_run": "227 assertions across 11 files passed with process inspection permitted; initial sandbox-stalled run stopped with exit 143 and is not a pass", + "generated_reference": "current", + "json_validation": "passed", + "diff_check": "passed", + "full_core_suite": "not rerun for this planning checkpoint; new focused R2 regressions are known to fail" + }, + "remaining": [ + "Repair native parser and distinguish requested settings from reported identity", + "Add strict evidence-import and actual-model independent-review public regressions", + "Retain safe failed native evidence in durable receipts and test recovery/replay", + "Run the repaired focused and full offline gates plus bounded independent patch review", + "Obtain a corrected live Claude-to-different-model-Codex receipt after normal Claude login" + ], + "completion_claim": "R2 remains in progress. The earlier 330-test R1 green gate predates this failing test file; it is not a current-tree pass." +} diff --git a/test/core/test_claude_identity.py b/test/core/test_claude_identity.py new file mode 100644 index 0000000..37f19fc --- /dev/null +++ b/test/core/test_claude_identity.py @@ -0,0 +1,311 @@ +"""Native Claude identity conformance without provider or authentication calls.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import shlex +import sys +import tempfile +import unittest +from unittest.mock import patch + +CORE = Path(__file__).resolve().parents[2] / "plugin" / "core" +sys.path.insert(0, str(CORE / "src")) + +from devsquad.claude_delivery_worker import ( + freeze_claude_implementer, + run as run_claude_implementer, +) +from devsquad.contracts import ContractError + + +class ClaudeImplementationIdentityTest(unittest.TestCase): + MODEL = "claude-sonnet-4-6" + + def setUp(self): + temporary = tempfile.TemporaryDirectory() + self.addCleanup(temporary.cleanup) + self.root = Path(temporary.name) + self.workspace = self.root / "workspace" + self.workspace.mkdir() + self.output = self.root / "provider-result.json" + self.binary = self.root / "claude" + self.binary.write_text( + "#!/bin/sh\n" + "if [ \"$1\" = \"--version\" ]; then\n" + " printf '%s\\n' '2.1.220 (Claude Code)'\n" + " exit 0\n" + "fi\n" + f"/bin/cat {shlex.quote(str(self.output))}\n" + ) + self.binary.chmod(0o700) + + @staticmethod + def model_usage(): + # Native modelUsage keys identify the model; these values are usage, + # not a requested configuration or an effective-effort observation. + return { + "inputTokens": 12, + "outputTokens": 7, + "cacheReadInputTokens": 0, + "cacheCreationInputTokens": 0, + "webSearchRequests": 0, + "costUSD": 0.0, + "contextWindow": 200000, + "maxOutputTokens": 64000, + } + + def document(self): + return { + "type": "result", + "subtype": "success", + "is_error": False, + "result": "Applied the bounded implementation.", + "session_id": "session-identity-fixture", + "usage": {"input_tokens": 12, "output_tokens": 7}, + "modelUsage": {self.MODEL: self.model_usage()}, + } + + def snapshot(self, model=None): + profile = { + "id": "claude-implementer", + "harness": "claude", + "model_family": "claude-sonnet", + "model_id": model or self.MODEL, + "effort": {"value": "high", "transport": "native"}, + "required_tools": ["read", "write"], + "permission_policy": "workspace_write", + "account_pool_id": "claude-subscription", + "billing_mode": "subscription", + "quality_status": "proven", + "evidence_refs": ["identity-fixture"], + } + selected = { + "reference": {"kind": "profile", "id": profile["id"]}, + "binding": None, + "profile_id": profile["id"], + "profile_sha256": hashlib.sha256(json.dumps( + profile, sort_keys=True, separators=(",", ":"), + ).encode()).hexdigest(), + "profile": profile, + } + baseline = "a" * 40 + task = { + "schema_version": 1, + "project": { + "repo_path": str(self.workspace), + "base_ref": baseline, + "target_ref": baseline, + }, + "workflow": "issue-delivery", + "goal": "Implement the identity fixture.", + "task_class": "identity-fixture", + "acceptance": [{ + "id": "implemented", + "description": "Implementation is complete.", + "evidence_kind": "review", + }], + "checks": [], + "scope": {"read_paths": ["src"], "write_paths": ["src"]}, + "lead": {"mode": "host"}, + "routing": { + "profiles_file": "devsquad/profiles.json", + "policy_file": "devsquad/policy.json", + }, + "budget": { + "wall_seconds": 5, + "max_worker_invocations": 1, + "max_revisions": 0, + "max_fallbacks_per_step": 0, + }, + "origin": {"surface": "test"}, + } + with patch.dict(os.environ, {"PATH": str(self.root)}): + adapter = freeze_claude_implementer(selected) + return { + "task": task, + "delivery_workspace": { + "path": str(self.workspace), + "baseline_oid": baseline, + "write_scope": task["scope"]["write_paths"], + }, + "routing": { + "roles": { + "implementer": {"selected": selected, "fallbacks": []}, + }, + }, + "implementation_adapter": adapter, + "implementation_adapters": {profile["id"]: adapter}, + } + + def run_document(self, document, *, requested_model=None): + self.output.write_text(json.dumps(document) + "\n") + return run_claude_implementer(self.snapshot(requested_model)) + + def test_exact_native_model_is_observed_but_effective_effort_is_unknown(self): + evidence = self.run_document(self.document()) + attempt = evidence["attempt"] + observed = attempt["observed_identity"] + self.assertEqual(observed["model_id"], self.MODEL) + self.assertEqual(observed["verification"], "verified") + self.assertIsNone(observed["effort"]) + self.assertIsNone(observed["backing_revision"]) + self.assertEqual( + attempt["selected_profile"]["profile"]["effort"]["value"], "high", + ) + self.assertEqual( + attempt["native_ids"]["session_id"], "session-identity-fixture", + ) + self.assertEqual(attempt["usage"]["total_tokens"], 19) + + def test_unexpected_model_cannot_be_verified_as_requested_model(self): + document = self.document() + document["modelUsage"] = {"claude-opus-4-6": self.model_usage()} + with self.assertRaises(ContractError): + self.run_document(document) + + def test_missing_model_usage_cannot_be_verified_from_requested_argv(self): + document = self.document() + del document["modelUsage"] + with self.assertRaises(ContractError): + self.run_document(document) + + def test_empty_or_malformed_model_usage_cannot_verify_identity(self): + values = [ + None, {}, [], "invalid", {"": self.model_usage()}, + {self.MODEL: None}, {self.MODEL: []}, {self.MODEL: "invalid"}, + {self.MODEL: {}}, + ] + for value in values: + with self.subTest(model_usage=value): + document = self.document() + document["modelUsage"] = value + with self.assertRaises(ContractError): + self.run_document(document) + + def test_malformed_native_counters_cannot_verify_identity(self): + for field in ("inputTokens", "outputTokens"): + for invalid in (None, -1, True, "12", 1.5): + with self.subTest(field=field, invalid=invalid): + document = self.document() + document["modelUsage"][self.MODEL][field] = invalid + with self.assertRaises(ContractError): + self.run_document(document) + with self.subTest(missing=field): + document = self.document() + del document["modelUsage"][self.MODEL][field] + with self.assertRaises(ContractError): + self.run_document(document) + + def test_multiple_model_entries_do_not_identify_a_unique_writer(self): + document = self.document() + document["modelUsage"]["claude-haiku-4-5-20251001"] = self.model_usage() + with self.assertRaises(ContractError): + self.run_document(document) + + def test_family_alias_resolves_only_to_reported_concrete_model(self): + evidence = self.run_document(self.document(), requested_model="sonnet") + attempt = evidence["attempt"] + observed = attempt["observed_identity"] + self.assertEqual(observed["model_id"], self.MODEL) + self.assertEqual(observed["verification"], "verified") + self.assertIsNone(observed["effort"]) + self.assertIsNone(observed["backing_revision"]) + self.assertEqual( + attempt["selected_profile"]["profile"]["model_id"], "sonnet", + ) + + def test_alias_cannot_resolve_to_a_different_model_family(self): + document = self.document() + document["modelUsage"] = {"claude-opus-4-6": self.model_usage()} + with self.assertRaises(ContractError): + self.run_document(document, requested_model="sonnet") + + def test_supported_family_aliases_resolve_without_a_baked_in_revision(self): + for alias, concrete in ( + ("sonnet", "claude-sonnet-4-6"), + ("opus", "claude-opus-4-6"), + ("haiku", "claude-haiku-4-5-20251001"), + ): + with self.subTest(alias=alias, concrete=concrete): + document = self.document() + document["modelUsage"] = {concrete: self.model_usage()} + evidence = self.run_document(document, requested_model=alias) + observed = evidence["attempt"]["observed_identity"] + self.assertEqual(observed["model_id"], concrete) + self.assertIsNone(observed["backing_revision"]) + + def test_reported_alias_is_not_concrete_execution_identity(self): + document = self.document() + document["modelUsage"] = {"sonnet": self.model_usage()} + with self.assertRaises(ContractError): + self.run_document(document, requested_model="sonnet") + + def test_pricing_canonical_model_does_not_replace_native_model_key(self): + document = self.document() + document["modelUsage"][self.MODEL]["canonicalModel"] = "pricing-only-model" + evidence = self.run_document(document) + self.assertEqual(evidence["attempt"]["observed_identity"]["model_id"], self.MODEL) + + def test_pricing_canonical_model_cannot_hide_an_unexpected_native_model(self): + document = self.document() + usage = self.model_usage() + usage["canonicalModel"] = self.MODEL + document["modelUsage"] = {"unexpected-serving-model": usage} + with self.assertRaises(ContractError): + self.run_document(document) + + def test_top_level_model_cannot_override_native_model_usage(self): + document = self.document() + document["model"] = self.MODEL + evidence = self.run_document(document) + self.assertEqual(evidence["attempt"]["observed_identity"]["model_id"], self.MODEL) + document["model"] = "claude-opus-4-6" + with self.assertRaises(ContractError): + self.run_document(document) + + def test_only_a_success_result_envelope_can_supply_identity(self): + changes = [ + {"type": "assistant"}, + {"subtype": "error_during_execution"}, + {"is_error": True}, + {"is_error": "false"}, + {"result": ""}, + {"result": None}, + ] + for change in changes: + with self.subTest(change=change): + document = {**self.document(), **change} + with self.assertRaises(ContractError): + self.run_document(document) + for missing in ("type", "subtype", "is_error", "result"): + with self.subTest(missing=missing): + document = self.document() + del document[missing] + with self.assertRaises(ContractError): + self.run_document(document) + + def test_missing_or_invalid_native_session_cannot_supply_identity(self): + for session in (None, "", " ", 42, [], "s" * 10000): + with self.subTest(session=session): + document = self.document() + document["session_id"] = session + with self.assertRaises(ContractError): + self.run_document(document) + document = self.document() + del document["session_id"] + with self.assertRaises(ContractError): + self.run_document(document) + + def test_non_object_native_json_fails_as_contract_error(self): + for document in (None, [], "success", 42): + with self.subTest(document=document): + with self.assertRaises(ContractError): + self.run_document(document) + + +if __name__ == "__main__": + unittest.main() From ef988897e2dda495482dd26e7c583bc428086ba0 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 00:07:35 -0700 Subject: [PATCH 134/197] WIP checkpoint: R2 identity repair preserved; final integration gate pending (2026-10-01 00:07) --- docs/plans/engineering-team/CONTRACTS.md | 22 + docs/plans/engineering-team/RESUME.md | 40 +- docs/plans/engineering-team/backlog.json | 2 +- .../src/devsquad/claude_delivery_worker.py | 101 ++-- plugin/core/src/devsquad/claude_identity.py | 283 +++++++++++ plugin/core/src/devsquad/reports.py | 3 + plugin/core/src/devsquad/service.py | 8 +- plugin/core/src/devsquad/supervisor.py | 51 +- plugin/core/src/devsquad/workflows.py | 53 +- test/core/test_delivery_identity_runtime.py | 462 ++++++++++++++++++ test/core/test_delivery_workflow.py | 3 +- .../test_implementation_identity_evidence.py | 225 +++++++++ 12 files changed, 1164 insertions(+), 89 deletions(-) create mode 100644 plugin/core/src/devsquad/claude_identity.py create mode 100644 test/core/test_delivery_identity_runtime.py create mode 100644 test/core/test_implementation_identity_evidence.py diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index 9fcb2a3..fad629f 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -81,6 +81,28 @@ same versioned binding shape and affects new runs only. Install-time discovery reports supported values and evidence (`documented`, `probed`, `unavailable`, `unknown`) with `checked_at`, CLI version and toolset hash. Selecting a known catalog entry verifies that it exists, not that the invocation used it: attempts retain separate `requested` and `observed` fields. Manually verified mappings may establish identity for a versioned harness; silent model fallback must never be labelled confirmed. +Claude implementation evidence v2 keeps the frozen requested profile separate +from `observed_identity`. The tested native result must have a success envelope, +bounded session ID and a single concrete `modelUsage` entry. Its model key is +the reported serving identity; `canonicalModel` is pricing metadata only. +Tested family aliases may resolve to a reported concrete member of that family, +without hardcoded current revisions. Multiple entries cannot identify a unique +writer and fail closed. Effective effort and backing revision remain null when +the result does not report them; `verification_scope: reported_model` does not +claim that these unknown settings were verified. Native session, typed usage, +alias resolution and the normalized result evidence are retained and validated +again on coordinator import, before candidate finalization. + +New delivery review imports and acceptance require verified different reported +model IDs, regardless of requested aliases, family labels or harness names. +Unknown or mixed native/fixture identities cannot establish independence; +explicit all-fixture runs test orchestration only. Legacy native implementation +v1 receipts remain readable and exact terminal decisions replayable, but cannot +authorize new acceptance. Rejected native results retain unverified, bounded +model/session/usage diagnostics, native-output hash/size and durable stream +artifact references across failure, fallback and cancellation. Provider prose +and arbitrary result fields are not copied into those diagnostic projections. + ### Automatic selection and manual overrides Default selection is automatic among policy-eligible profiles. The host supplies task requirements; the deterministic router chooses the model/effort/tool profile without another planning-model call. Within the selected toolbox, the worker chooses individual tool calls. Discovery can enumerate supported configurations; initial quality preferences and account-pool mappings still require evidence and operator setup. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index c9f17b1..615c109 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -2,25 +2,35 @@ This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. -## Current position — September 29, 2026 +## Current position — October 1, 2026 ### Review correction and next action -Latest handoff is a **plan plus preserved failing R2 regressions**, not an R2 -repair. Production remains at R1 commit `6848f11`. -`test/core/test_claude_identity.py` now contains 16 fake-CLI tests; the focused -baseline ran in 3.162 seconds with 39 assertion/subtest failures and 4 errors -(non-object JSON raises `AttributeError`). No live provider call or new -production change occurred. The current test tree is knowingly red; the -330-test green result below predates these new regressions. See +R2 implementation is preserved after `3612942`: strict native result parsing, +v2 import evidence, actual-model independence and durable failed diagnostics. +The 16 original worker regressions, 12 import/parser regressions, 19 existing +delivery tests and nine public native delivery/fallback/cancellation/legacy +tests have passed focused runs. Independent review identified two additional +gaps (actual durable-attempt profile binding and exact native-byte retention); +both were reproduced, repaired and independently rechecked with four passing +targeted regressions and no additional actionable finding. + +The first integrated gate before those last two fixes passed 363 tests with +two optional-SDK skips. The subsequent 367-test gate completed with **six +failures and two skips**, not a pass. Four failures show budget expiration or +15–17 minute UTC jumps during a 231-second monotonic suite. On October 1, all +six failed cases passed unchanged in 18.998 seconds; per-test UTC and monotonic +elapsed measurements agreed. This supports an environmental timing explanation, +but does not erase the failed gate. A clean complete rerun remains the next +verification step; do not weaken budget or identity enforcement. The recurring +SQLite cleanup warning remains assigned to R5. **R2 is not closed yet.** + +No live provider call, installed refresh or global provider-setting change +occurred. The original red baseline is retained in [R2 baseline evidence](evidence/R2-identity-red-baseline-2026-09-29.json). -This handoff's Bash gate passed all 227 assertions across 11 files with -process inspection permitted; the first sandbox-stalled run was stopped and -is not a pass. Generated-reference, JSON and whitespace checks passed. No -complete core rerun was needed for this plan-only checkpoint; R2 remains red. -Sol's next slice is R2a parser/worker, then R2b strict import/actual-model -independence and R2c failed-receipt/replay coverage, as detailed in the -[updated execution plan](SOL-REVIEW-FOLLOWUP.md#ready-to-execute-handoff-for-sol). +Once R2's final gate is green and recorded, continue R3 rather than retrying +blocked authentication. R3's two-outcome/three-case promotion bug was reproduced +again without modifying production code; it still needs provenance repair. R1's source repair is verified after `399d93d`: the public regression first reproduced four unsafe mutation paths (review/delivery × host/headless). The diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 375af96..16f7fda 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -23,7 +23,7 @@ }, "review_work_packages": [ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, - {"id": "R2", "title": "Observed Claude execution identity", "status": "in_progress", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-identity-red-baseline-2026-09-29.json", "checkpoint": "16 offline worker regressions saved; focused run reports 39 assertion/subtest failures and 4 errors against unchanged R1 production source. No R2 repair or green gate yet. Next: parser/worker, strict import/actual-model independence, then durable failure and replay coverage."}, + {"id": "R2", "title": "Observed Claude execution identity", "status": "in_progress", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-identity-red-baseline-2026-09-29.json", "checkpoint": "Source repair and public identity/failure/fallback/replay regressions implemented; two independent-review findings fixed and retested. Last full gate: 367 tests, six failures, two optional-SDK skips with observed wall-clock budget jumps. All six failures passed unchanged on October 1 with stable UTC/monotonic timing. Final complete gate and evidence checkpoint remain; installed/live proof stays open."}, {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "pending", "milestones": ["M6"], "depends_on": [], "items": ["F3"]}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, diff --git a/plugin/core/src/devsquad/claude_delivery_worker.py b/plugin/core/src/devsquad/claude_delivery_worker.py index 8a4eeb2..437b7aa 100644 --- a/plugin/core/src/devsquad/claude_delivery_worker.py +++ b/plugin/core/src/devsquad/claude_delivery_worker.py @@ -11,6 +11,10 @@ from typing import Any from .contracts import CapabilityUnavailable, ContractError, ProfileUnsupported +from .claude_identity import ( + ClaudeResultError, failure_diagnostics, failure_envelope, native_result, + observed_identity, strict_json, +) from .store import canonical_json from .workflows import build_implementation_prompt, make_implementation_evidence @@ -122,41 +126,6 @@ def _validated(snapshot: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any] return adapter, profile -def _structured_result(payload: str) -> tuple[str, str, dict[str, Any]]: - try: - document = json.loads(payload) - except json.JSONDecodeError as exc: - raise ContractError("Claude implementation output is not JSON") from exc - if (not isinstance(document, dict) or document.get("type") != "result" - or document.get("is_error") is True - or not isinstance(document.get("result"), str) - or not document["result"].strip()): - raise ContractError("Claude implementation output has no successful result") - session_id = document.get("session_id") - if not isinstance(session_id, str) or not session_id: - raise ContractError("Claude implementation output has no session id") - raw_usage = document.get("usage") - usage = { - "input_tokens": None, - "output_tokens": None, - "total_tokens": None, - "source": "unavailable", - } - if isinstance(raw_usage, dict): - input_tokens = raw_usage.get("input_tokens") - output_tokens = raw_usage.get("output_tokens") - if all(type(value) is int and value >= 0 for value in ( - input_tokens, output_tokens, - )): - usage = { - "input_tokens": input_tokens, - "output_tokens": output_tokens, - "total_tokens": input_tokens + output_tokens, - "source": "native_reported", - } - return document["result"].strip(), session_id, usage - - def run(snapshot: dict[str, Any]) -> dict[str, Any]: if not isinstance(snapshot, dict): raise ContractError("delivery snapshot must be an object") @@ -175,7 +144,7 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: [str(binary), "--version"], text=True, capture_output=True, timeout=3, check=False, ) - except (OSError, subprocess.TimeoutExpired) as exc: + except (OSError, subprocess.TimeoutExpired, UnicodeError) as exc: raise CapabilityUnavailable("Claude version could not be re-observed") from exc if version.returncode != 0 or version.stdout.strip() != adapter["harness_version"]: raise CapabilityUnavailable("Claude version changed after preflight") @@ -200,7 +169,7 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: argv, cwd=workspace, env={**os.environ, "DEVSQUAD_WORKER": "1"}, - text=True, + text=False, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=timeout_seconds, @@ -212,15 +181,17 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: argv, 124, exc.stdout or "", exc.stderr or "", ) timed_out = True - stdout = completed.stdout.decode("utf-8", "replace") if isinstance( - completed.stdout, bytes - ) else completed.stdout + # Preserve exact native bytes, including invalid UTF-8 and CRLF. Only + # stderr classification uses replacement decoding; stdout is parsed strictly. + stdout = completed.stdout stderr = completed.stderr.decode("utf-8", "replace") if isinstance( completed.stderr, bytes ) else completed.stderr - if (len(stdout.encode()) > MAX_CLAUDE_OUTPUT_BYTES + if (len(stdout.encode() if isinstance(stdout, str) else stdout) > MAX_CLAUDE_OUTPUT_BYTES or len(stderr.encode()) > MAX_CLAUDE_OUTPUT_BYTES): - raise ContractError("Claude implementation output exceeds its byte limit") + raise ClaudeResultError("CLI_ERROR", failure_diagnostics( + stdout, None, "output_limit", + )) import re error_code = next(( code for code, pattern in adapter["error_patterns"].items() @@ -230,15 +201,15 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: error_code = "TIMEOUT" elif completed.returncode != 0 and error_code is None: error_code = "CLI_ERROR" + provider_document = None try: - summary, session_id, usage = _structured_result(stdout) + provider_document = strict_json(stdout) + summary, native = native_result(provider_document) + observed = observed_identity(native, adapter, profile) except ContractError as exc: - try: - provider_document = json.loads(stdout) - except json.JSONDecodeError: - provider_document = {} + safe_document = provider_document if isinstance(provider_document, dict) else {} provider_text = str( - provider_document.get("result") or provider_document.get("error") or "" + safe_document.get("result") or safe_document.get("error") or "" ) error_code = error_code or next(( code for code, pattern in adapter["error_patterns"].items() @@ -248,27 +219,20 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: adapter["denied_pattern"], provider_text, re.IGNORECASE, ): error_code = "CLI_ERROR" - raise ContractError( - f"{error_code or 'CLI_ERROR'}: Claude implementation failed" - ) from exc + raise ClaudeResultError(error_code or "CLI_ERROR", failure_diagnostics( + stdout, provider_document, "native_result_invalid", + )) from exc if error_code is not None: - raise ContractError(f"{error_code}: Claude implementation failed") - observed = { - "harness": "claude", - "harness_version": adapter["harness_version"], - "model_provider": adapter["model_provider"], - "model_id": model, - "effort": effort, - "permission_policy": "workspace_write", - "verification": "verified", - } + raise ClaudeResultError(error_code, failure_diagnostics( + stdout, provider_document, "execution_failed", + )) return make_implementation_evidence( snapshot, summary, observed_identity=observed, - native_ids={"session_id": session_id}, + native_ids={"session_id": native["session_id"]}, native_model_requests=None, - usage=usage, + usage=native["usage"], ) @@ -280,7 +244,16 @@ def main() -> int: snapshot = json.loads(payload.decode("utf-8")) except (UnicodeDecodeError, json.JSONDecodeError) as exc: raise ContractError("delivery snapshot is not valid UTF-8 JSON") from exc - sys.stdout.write(canonical_json(run(snapshot)) + "\n") + try: + result = run(snapshot) + except ClaudeResultError as exc: + selected = snapshot["routing"]["roles"]["implementer"]["selected"] + sys.stdout.write(canonical_json(failure_envelope( + exc, selected["profile_sha256"], + )) + "\n") + sys.stderr.write(str(exc) + "\n") + return 1 + sys.stdout.write(canonical_json(result) + "\n") return 0 diff --git a/plugin/core/src/devsquad/claude_identity.py b/plugin/core/src/devsquad/claude_identity.py new file mode 100644 index 0000000..54a7b40 --- /dev/null +++ b/plugin/core/src/devsquad/claude_identity.py @@ -0,0 +1,283 @@ +"""Bounded Claude result evidence; requested settings are never observations. + +The tested CLI result attributes usage to model IDs, not individual writer +messages. Only a single concrete reported model can establish writer identity. +Pricing aliases are metadata. Effective effort and serving revision are unknown. +""" + +from __future__ import annotations + +import hashlib +import json +import math +import re +from typing import Any + +from .contracts import ContractError + + +MAX_OUTPUT_BYTES = 2 * 1024 * 1024 +MAX_COUNTER = 2 ** 63 - 1 +_MODEL = re.compile(r"[A-Za-z0-9][A-Za-z0-9._:/-]{0,255}\Z") +_SESSION = re.compile(r"[A-Za-z0-9][A-Za-z0-9._:-]{0,499}\Z") +_COUNTERS = { + "inputTokens", "outputTokens", "cacheReadInputTokens", + "cacheCreationInputTokens", "webSearchRequests", "contextWindow", + "maxOutputTokens", +} +_ALIASES = {"sonnet", "opus", "haiku"} +_NATIVE_FIELDS = { + "schema_version", "type", "subtype", "is_error", "session_id", + "model_usage", "top_level_model", "usage", +} +ERROR_CODES = {"AUTH_ERROR", "RATE_LIMITED", "TIMEOUT", "CLI_ERROR"} + + +def _exact(value: Any, fields: set[str]) -> dict[str, Any]: + if not isinstance(value, dict) or set(value) != fields: + raise ContractError("Claude evidence fields are invalid") + return value + + +def _model(value: Any) -> str: + if not isinstance(value, str) or not _MODEL.fullmatch(value): + raise ContractError("Claude reported model is invalid") + return value + + +def _session(value: Any) -> str: + if not isinstance(value, str) or not _SESSION.fullmatch(value): + raise ContractError("Claude result session is invalid") + return value + + +def _counter(value: Any) -> int: + if type(value) is not int or not 0 <= value <= MAX_COUNTER: + raise ContractError("Claude reported counter is invalid") + return value + + +def strict_json(payload: bytes | str) -> Any: + def pairs(items): + result = {} + for key, value in items: + if key in result: + raise ValueError("duplicate key") + result[key] = value + return result + + def nonfinite(value): + raise ValueError("non-finite number") + + def finite_float(value): + result = float(value) + if not math.isfinite(result): + raise ValueError("non-finite number") + return result + + try: + encoded = payload.encode("utf-8") if isinstance(payload, str) else payload + if len(encoded) > MAX_OUTPUT_BYTES: + raise ValueError("output limit") + return json.loads(encoded.decode("utf-8"), object_pairs_hook=pairs, + parse_constant=nonfinite, parse_float=finite_float) + except (UnicodeError, ValueError, TypeError, RecursionError) as exc: + raise ContractError("Claude result is not bounded strict UTF-8 JSON") from exc + + +def unknown_usage() -> dict[str, Any]: + return {"input_tokens": None, "output_tokens": None, + "total_tokens": None, "source": "unavailable"} + + +def reported_usage(raw: Any) -> dict[str, Any]: + if not isinstance(raw, dict): + return unknown_usage() + try: + incoming = _counter(raw.get("input_tokens")) + outgoing = _counter(raw.get("output_tokens")) + total = _counter(incoming + outgoing) + except ContractError: + return unknown_usage() + return {"input_tokens": incoming, "output_tokens": outgoing, + "total_tokens": total, "source": "native_reported"} + + +def _usage_evidence(value: Any) -> dict[str, Any]: + _exact(value, set(unknown_usage())) + expected = reported_usage(value) + if expected != value: + raise ContractError("Claude usage evidence is inconsistent") + return expected + + +def model_usage(raw: Any) -> dict[str, Any]: + if not isinstance(raw, dict) or not 1 <= len(raw) <= 32: + raise ContractError("Claude result has no bounded model usage") + result = {} + for model, entry in raw.items(): + _model(model) + if not isinstance(entry, dict): + raise ContractError("Claude model usage is invalid") + normalized = {key: _counter(entry[key]) for key in + ("inputTokens", "outputTokens") if key in entry} + if len(normalized) != 2: + raise ContractError("Claude model usage counters are missing") + for key in _COUNTERS & entry.keys(): + normalized[key] = _counter(entry[key]) + if "costUSD" in entry: + cost = entry["costUSD"] + if (type(cost) not in (int, float) or not 0 <= cost <= MAX_COUNTER + or not math.isfinite(cost)): + raise ContractError("Claude model cost is invalid") + normalized["costUSD"] = cost + for key in ("canonicalModel", "provider"): + if key in entry: + normalized[key] = _model(entry[key]) + result[model] = normalized + return result + + +def native_result(document: Any) -> tuple[str, dict[str, Any]]: + if (not isinstance(document, dict) or document.get("type") != "result" + or document.get("subtype") != "success" + or document.get("is_error") is not False + or not isinstance(document.get("result"), str) + or not document["result"].strip()): + raise ContractError("Claude implementation has no successful result") + summary = document["result"].strip() + if len(summary) > 20_000: + raise ContractError("Claude summary exceeds its limit") + return summary, { + "schema_version": 1, "type": "result", "subtype": "success", + "is_error": False, "session_id": _session(document.get("session_id")), + "model_usage": model_usage(document.get("modelUsage")), + "top_level_model": _model(document["model"]) if "model" in document else None, + "usage": reported_usage(document.get("usage")), + } + + +def observed_identity(native: Any, adapter: dict[str, Any], + profile: dict[str, Any]) -> dict[str, Any]: + _exact(native, _NATIVE_FIELDS) + if (type(native["schema_version"]) is not int or native["schema_version"] != 1 + or native["type"] != "result" or native["subtype"] != "success" + or native["is_error"] is not False): + raise ContractError("Claude native identity envelope is invalid") + _session(native["session_id"]) + _usage_evidence(native["usage"]) + models = model_usage(native["model_usage"]) + if models != native["model_usage"] or len(models) != 1: + raise ContractError("Claude result cannot identify a unique writer model") + model = next(iter(models)) + if model in _ALIASES: + raise ContractError("Claude reported identity is an unresolved alias") + requested = _model(profile.get("model_id")) + alias = requested in _ALIASES + if (alias and not model.startswith(f"claude-{requested}-")) or ( + not alias and requested != model): + raise ContractError("Claude reported model does not match the requested profile") + top = native["top_level_model"] + if top is not None and _model(top) != model: + raise ContractError("Claude result contains contradictory model identity") + if (adapter.get("harness") != "claude" + or adapter.get("model_provider") != "anthropic" + or not isinstance(adapter.get("harness_version"), str) + or not adapter["harness_version"] + or profile.get("harness") != "claude" + or profile.get("permission_policy") != "workspace_write"): + raise ContractError("Claude identity does not match the frozen adapter") + provider = models[model].get("provider") + if provider is not None and provider != adapter["model_provider"]: + raise ContractError("Claude reported provider contradicts the frozen adapter") + return { + "harness": "claude", "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], "model_id": model, + "effort": None, "backing_revision": None, + "permission_policy": "workspace_write", "verification": "verified", + "verification_scope": "reported_model", + "model_source": "claude.result.modelUsage", + "alias_resolution": {"requested": requested, "reported": model} if alias else None, + "native_evidence": native, + } + + +def validate_observation(value: Any, adapter: dict[str, Any], + profile: dict[str, Any], ids: Any, usage: Any) -> None: + if not isinstance(value, dict): + raise ContractError("Claude observed identity is missing") + native = value.get("native_evidence") + expected = observed_identity(native, adapter, profile) + # Canonical JSON comparison distinguishes booleans from integers. + if json.dumps(value, sort_keys=True) != json.dumps(expected, sort_keys=True): + raise ContractError("Claude observed identity differs from native evidence") + if ids != {"session_id": native["session_id"]} or usage != native["usage"]: + raise ContractError("Claude session or usage differs from native evidence") + + +def failure_diagnostics(payload: bytes | str, document: Any, reason: str) -> dict[str, Any]: + """Keep only bounded typed native fields, never provider prose or stderr.""" + document = document if isinstance(document, dict) else {} + session = document.get("session_id") + try: + _session(session) + except ContractError: + session = None + models = {} + raw = document.get("modelUsage") + if isinstance(raw, dict) and len(raw) <= 32: + for key, value in raw.items(): + try: + models.update(model_usage({key: value})) + except (ContractError, OverflowError): + continue + encoded = payload.encode("utf-8") if isinstance(payload, str) else payload + return { + "schema_version": 1, "identity_status": "unverified", "reason": reason, + "output_sha256": hashlib.sha256(encoded).hexdigest(), + "output_bytes": len(encoded), "session_id": session, + "model_usage": models, "usage": reported_usage(document.get("usage")), + } + + +class ClaudeResultError(ContractError): + def __init__(self, code: str, diagnostics: dict[str, Any]): + super().__init__(f"{code}: Claude implementation failed") + self.code = code + self.diagnostics = diagnostics + + +def failure_envelope(error: ClaudeResultError, profile_sha256: str) -> dict[str, Any]: + return {"schema_version": 1, "type": "claude_implementation_failure", + "profile_sha256": profile_sha256, "error": error.code, + "native_diagnostics": error.diagnostics} + + +def validate_failure(value: Any, profile_sha256: str) -> dict[str, Any]: + _exact(value, {"schema_version", "type", "profile_sha256", "error", + "native_diagnostics"}) + if (type(value["schema_version"]) is not int or value["schema_version"] != 1 + or value["type"] != "claude_implementation_failure" + or value["profile_sha256"] != profile_sha256 + or not isinstance(value["error"], str) + or value["error"] not in ERROR_CODES): + raise ContractError("Claude failed attempt identity is invalid") + data = _exact(value["native_diagnostics"], { + "schema_version", "identity_status", "reason", "output_sha256", + "output_bytes", "session_id", "model_usage", "usage", + }) + if (type(data["schema_version"]) is not int or data["schema_version"] != 1 + or data["identity_status"] != "unverified" + or not isinstance(data["reason"], str) + or data["reason"] not in {"native_result_invalid", "execution_failed", "output_limit"} + or not isinstance(data["output_sha256"], str) + or not re.fullmatch(r"[0-9a-f]{64}", data["output_sha256"])): + raise ContractError("Claude failure diagnostics are invalid") + _counter(data["output_bytes"]) + if data["session_id"] is not None: + _session(data["session_id"]) + if data["model_usage"] != {}: + if model_usage(data["model_usage"]) != data["model_usage"]: + raise ContractError("Claude failed model usage is invalid") + _usage_evidence(data["usage"]) + return value diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index 2e29188..f406869 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -276,6 +276,9 @@ def build_early_terminal_reports( "output_artifacts": projected, "error": error, } + if "native_diagnostics" in attempt: + attempt_projection["native_diagnostics"] = attempt["native_diagnostics"] + attempt_projection["usage"] = attempt["native_diagnostics"]["usage"] criteria = [{ "id": criterion.get("id"), "description": criterion.get("description"), diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index b086825..b2f21d0 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -1355,6 +1355,8 @@ def _review_gate( validate_handoff_decision_evidence(decision, packet) if decision.get("disposition") == "accept": require_check_integrity(packet["checks"]) + from .workflows import require_independent_delivery_review + require_independent_delivery_review(snapshot, packet) revisions_used = sum( entry["decision"]["disposition"] == "revise" for entry in store.branch_review_history(run_id) @@ -1433,12 +1435,14 @@ def _failed_fallback_attempts( "observed_identity": None, "worker_invocations": 1, "native_model_requests": None, - "usage": { + "usage": error.get("native_diagnostics", {}).get("usage", { "input_tokens": None, "output_tokens": None, "total_tokens": None, "source": "unavailable", - }, + }), + **({"native_diagnostics": error["native_diagnostics"]} + if "native_diagnostics" in error else {}), "error": error, }) return failures diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index ac848b8..08c7710 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -18,12 +18,14 @@ import json from .contracts import ContractError, LaunchSpec +from .claude_identity import strict_json, validate_failure from .reports import build_early_terminal_reports from .store import AttemptReservation, ConflictError, Store, canonical_json from .workflows import ( decode_branch_review_evidence, decode_headless_lead_evidence, review_mode, + require_independent_delivery_review, validate_implementation_evidence, validate_review_document, ) @@ -35,6 +37,26 @@ ) +def _require_attempt_profile( + evidence: dict[str, Any], snapshot: dict[str, Any], + attempt: dict[str, Any], role: str, +) -> None: + """A permitted fallback is not proof that this invocation used it.""" + try: + routed = snapshot["routing"]["roles"][role] + candidates = [routed["selected"], *routed["fallbacks"]] + index = attempt["profile_index"] + if type(index) is not int or not 0 <= index < len(candidates): + raise ContractError("durable attempt profile index is invalid") + selected = candidates[index] + if (selected["profile_id"] != attempt["profile_id"] + or canonical_json(evidence["attempt"]["selected_profile"]) + != canonical_json(selected)): + raise ContractError("imported evidence differs from the actual attempt profile") + except (KeyError, TypeError, IndexError) as exc: + raise ContractError("durable attempt profile binding is missing") from exc + + def _open_stdin_artifact(path: str) -> BinaryIO: candidate = Path(path) if candidate.is_symlink(): @@ -340,6 +362,7 @@ def _commit_review_handoff( stdout: bytes, ) -> str: evidence = decode_branch_review_evidence(stdout, snapshot) + _require_attempt_profile(evidence, snapshot, attempt, "reviewer") suffix = attempt["id"] documents = { f"review-{suffix}.json": evidence["review"], @@ -399,8 +422,9 @@ def _commit_delivery_candidate( stdout: bytes, ) -> str: evidence = validate_implementation_evidence( - json.loads(stdout.decode("utf-8")), snapshot, + strict_json(stdout), snapshot, ) + _require_attempt_profile(evidence, snapshot, attempt, "implementer") task = snapshot["task"] delivery = snapshot["delivery_workspace"] source_repo = Path(task["project"]["repo_path"]).resolve(strict=True) @@ -596,6 +620,20 @@ def import_durable(self, run_id: str) -> str: workflow_delivery = managed_workflow and workflow == "issue-delivery" role = attempt.get("role", "worker") semantic_error=None + native_failure = None + if role == "implementer" and workflow_delivery: + # Only normalized, profile-bound diagnostics may reach reports. + # Raw output remains a hash-bound private artifact, not identity. + try: + routed = snapshot["routing"]["roles"][role] + selected = [routed["selected"], *routed["fallbacks"]][attempt["profile_index"]] + adapter = snapshot.get("implementation_adapters", {}).get(selected["profile_id"]) + if isinstance(adapter, dict) and adapter.get("harness") == "claude": + native_failure = validate_failure( + strict_json(captures["stdout"]), selected["profile_sha256"], + ) + except (ContractError, KeyError, TypeError, IndexError): + pass if (role == "implementer" and workflow_delivery and not receipt["cancelled"] and not receipt["timed_out"] and receipt["returncode"] == 0): @@ -627,6 +665,9 @@ def import_durable(self, run_id: str) -> str: evidence = decode_headless_lead_evidence( captures["stdout"], snapshot, frozen_handoff, ) + _require_attempt_profile(evidence, snapshot, attempt, "lead") + if evidence["choice"]["disposition"] == "accept": + require_independent_delivery_review(snapshot, handoff.packet) content = (canonical_json(evidence) + "\n").encode() name = f"lead-attempt-{attempt['id']}.json" path,digest,size=self.store.finalize_artifact(run_id,name,content) @@ -705,6 +746,12 @@ def import_durable(self, run_id: str) -> str: "returncode": receipt["returncode"], } payload.update(report_error) + if native_failure is not None: + diagnostics = native_failure["native_diagnostics"] + metadata["native_diagnostics"] = diagnostics + if report_error is not None: + report_error["native_diagnostics"] = diagnostics + payload["native_diagnostics"] = diagnostics candidates = [] profile_index = None try: @@ -759,6 +806,8 @@ def import_durable(self, run_id: str) -> str: "returncode": receipt["returncode"], "cancelled": receipt["cancelled"], "timed_out": receipt["timed_out"], + **({"native_diagnostics": native_failure["native_diagnostics"]} + if native_failure is not None else {}), "selected_profile": ( candidates[profile_index] if type(profile_index) is int diff --git a/plugin/core/src/devsquad/workflows.py b/plugin/core/src/devsquad/workflows.py index a1fe301..57f395e 100644 --- a/plugin/core/src/devsquad/workflows.py +++ b/plugin/core/src/devsquad/workflows.py @@ -9,6 +9,7 @@ from typing import Any from .contracts import ContractError +from .claude_identity import validate_observation from .store import canonical_json from .validation import validate_task @@ -585,7 +586,7 @@ def validate_implementation_evidence( document = _exact(value, { "schema_version", "workflow", "baseline_oid", "summary", "attempt", }, "implementation evidence") - if document["schema_version"] != 1 or type(document["schema_version"]) is not int: + if type(document["schema_version"]) is not int or document["schema_version"] not in {1, 2}: raise ContractError("implementation evidence schema_version is invalid") if document["workflow"] != "issue-delivery": raise ContractError("implementation evidence workflow is invalid") @@ -624,10 +625,16 @@ def validate_implementation_evidence( selected["profile_id"] ) if adapter is None: - if attempt["observed_identity"] is not None or attempt["native_ids"] != {}: + if (document["schema_version"] != 1 + or attempt["observed_identity"] is not None or attempt["native_ids"] != {}): raise ContractError("fixture implementation cannot claim native identity") - elif not isinstance(attempt["observed_identity"], dict): - raise ContractError("native implementation identity is missing") + else: + if document["schema_version"] != 2: + raise ContractError("native implementation requires v2 reported identity evidence") + validate_observation( + attempt["observed_identity"], adapter, selected["profile"], + attempt["native_ids"], attempt["usage"], + ) if attempt["worker_invocations"] != 1 or type(attempt["worker_invocations"]) is not int: raise ContractError("implementation worker invocation accounting is invalid") native_requests = attempt["native_model_requests"] @@ -662,7 +669,7 @@ def make_implementation_evidence( ) -> dict[str, Any]: selected = snapshot["routing"]["roles"]["implementer"]["selected"] document = { - "schema_version": 1, + "schema_version": 2 if observed_identity is not None else 1, "workflow": "issue-delivery", "baseline_oid": snapshot["delivery_workspace"]["baseline_oid"], "summary": summary, @@ -1091,9 +1098,45 @@ def decode_branch_review_evidence( snapshot, ) require_check_integrity(evidence["checks"]) + require_independent_delivery_review(snapshot, evidence) return evidence +def require_independent_delivery_review( + snapshot: dict[str, Any], review: dict[str, Any], +) -> None: + """Gate new imports/acceptances; historical receipts remain readable.""" + if snapshot.get("task", {}).get("workflow") != "issue-delivery": + return + candidate = review.get("candidate_sha256") + iteration = next((item for item in snapshot.get("delivery_iterations", []) + if item.get("candidate", {}).get("candidate_sha256") == candidate), None) + if iteration is None: + raise ContractError("independent review has no saved implementation candidate") + implementation_snapshot = dict(snapshot) + implementation_snapshot.pop("revision_request", None) + if "revision_request" in iteration: + implementation_snapshot["revision_request"] = iteration["revision_request"] + implementation = validate_implementation_evidence( + iteration.get("implementation"), implementation_snapshot, + ) + writer = implementation["attempt"]["observed_identity"] + reviewer = review["attempt"]["observed_identity"] + # Explicit all-fixture runs exercise orchestration, never native identity. + if (writer is None and reviewer is None + and "internal_implementation_fixture" in snapshot + and "internal_review_fixture" in snapshot): + return + if (not isinstance(writer, dict) or not isinstance(reviewer, dict) + or writer.get("verification") != "verified" + or reviewer.get("verification") != "verified" + or not isinstance(reviewer.get("model_id"), str) + or reviewer["model_id"] in {"sonnet", "opus", "haiku"}): + raise ContractError("delivery requires verified independent native model identities") + if writer["model_id"].casefold() == reviewer["model_id"].casefold(): + raise ContractError("delivery reviewer model is not independent of the implementer") + + def require_check_integrity(checks: list[dict[str, Any]]) -> None: """Keep v1 receipts readable, but never use them for a new acceptance.""" if any(check.get("schema_version") != 2 for check in checks): diff --git a/test/core/test_delivery_identity_runtime.py b/test/core/test_delivery_identity_runtime.py new file mode 100644 index 0000000..d979e12 --- /dev/null +++ b/test/core/test_delivery_identity_runtime.py @@ -0,0 +1,462 @@ +"""Native execution identity gates through public durable delivery operations.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import shlex +import sys +import time +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.reports import TERMINAL_REPORT_NAMES +from devsquad.service import Service +from devsquad.store import Store, canonical_json +import test_delivery_workflow as delivery_fixtures + + +class PublicDeliveryIdentityTest(unittest.TestCase): + MODEL = "claude-sonnet-4-6" + FALLBACK_MODEL = "claude-opus-4-6" + + def setUp(self): + # Compose the existing fixture instead of inheriting all its test cases. + self.fixture = delivery_fixtures.DeliveryWorkspaceTest(methodName="runTest") + self.fixture.setUp() + self.addCleanup(self.fixture.doCleanups) + self.root, self.repo = self.fixture.root, self.fixture.repo + self.service = Service(self.fixture.runtime) + self.run_ids = [] + self.fallbacks = False + self.addCleanup(self.cancel_unfinished_runs) + self.output = self.root / "native-result.json" + fake_bin = self.root / "fake-bin" + fake_bin.mkdir() + self.fake_claude = fake_bin / "claude" + self.fake_claude.write_text( + "#!/bin/sh\n" + "if [ \"$1\" = \"--version\" ]; then\n" + " printf '%s\\n' '2.1.220 (Claude Code)'\n" + " exit 0\n" + "fi\n" + "printf '%s\\n' \"VALUE = 'fixed'\" > src/app.py\n" + f"/bin/cat {shlex.quote(str(self.output))}\n" + ) + self.fake_claude.chmod(0o700) + (fake_bin / "codex").symlink_to(ROOT / "test/core/fakes/codex_review_cli.py") + fake_auth = self.root / "native-auth.json" + fake_auth.write_text("{}\n") + fake_auth.chmod(0o600) + self.environment = {"PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}"} + # Only preflight's file lookup is patched. Both worker binaries and all + # durable service/import/review paths execute normally in child processes. + self.auth_patch = patch( + "devsquad.codex_review_worker._subscription_auth_file", + return_value=fake_auth, + ) + self.auth_patch.start() + self.addCleanup(self.auth_patch.stop) + + def cancel_unfinished_runs(self): + for run_id in self.run_ids: + if self.service.status(run_id)["state"] not in {"succeeded", "failed", "cancelled"}: + self.service.cancel(run_id) + + @staticmethod + def counters(): + return {"inputTokens": 12, "outputTokens": 7} + + def native_result(self): + return { + "type": "result", "subtype": "success", "is_error": False, + "result": "Applied the scoped native fixture implementation.", + "session_id": "native-delivery-session", + "usage": {"input_tokens": 12, "output_tokens": 7}, + "modelUsage": {self.MODEL: self.counters()}, + } + + def configure(self, *, requested_model=None, reviewer_model="gpt-fake-review", fallback=False): + profiles, policy = self.fixture.delivery_routing_documents() + self.fallbacks = fallback + profiles["profiles"] = [ + profile for profile in profiles["profiles"] + if fallback or profile["id"] != "fixture-implementer-fallback" + ] + by_id = {profile["id"]: profile for profile in profiles["profiles"]} + implementer = by_id["fixture-implementer"] + reviewer = by_id["fixture-reviewer"] + lead = by_id["fixture-lead"] + implementer.update({ + "harness": "claude", "model_family": "writer-family-label", + "model_id": requested_model or self.MODEL, + "effort": {"value": "high", "transport": "native"}, + }) + if fallback: + by_id["fixture-implementer-fallback"].update({ + "harness": "claude", "model_family": "fallback-family-label", + "model_id": self.FALLBACK_MODEL, + "effort": {"value": "high", "transport": "native"}, + }) + reviewer.update({ + "harness": "codex", "model_family": "distinct-reviewer-label", + "model_id": reviewer_model, + }) + lead.update({"harness": "codex", "model_id": "gpt-fake-lead"}) + if not fallback: + policy["roles"]["implementer"] = policy["roles"]["implementer"][:1] + for name, document in (("profiles", profiles), ("policy", policy)): + (self.repo / "devsquad" / f"{name}.json").write_text(json.dumps(document)) + self.fixture.git(self.repo, "add", "devsquad") + self.fixture.git(self.repo, "commit", "-qm", "native identity fixture") + self.fixture.baseline = self.fixture.git(self.repo, "rev-parse", "HEAD").strip() + self.fixture.source_refs = self.fixture.git(self.repo, "show-ref") + self.fixture.source_status = self.fixture.git(self.repo, "status", "--porcelain") + + def task(self, mode="host"): + task = self.fixture.delivery_task() + task["lead"] = {"mode": mode} + task["budget"]["wall_seconds"] = 60 + task["budget"]["max_fallbacks_per_step"] = int(self.fallbacks) + return task + + def start(self, document, key, *, mode="host"): + self.output.write_text(json.dumps(document)) + with patch.dict(os.environ, self.environment): + started = self.service.start(self.task(mode), key) + self.run_ids.append(started["run_id"]) + self.assertEqual(started["state"], "queued", started) + return started["run_id"] + + def wait(self, run_id, *, candidate=False): + deadline = time.monotonic() + 20 + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] in {"succeeded", "failed", "cancelled", "awaiting_host"}: + return status + if (candidate and status["state"] == "queued" + and status.get("next_action") == "resume_candidate_review"): + return status + time.sleep(0.05) + self.fail(f"identity run did not reach its next gate: {self.service.status(run_id)}") + + def snapshot(self, run_id): + store = Store(self.service.database, self.service.artifacts) + try: + return json.loads(store.run(run_id)["mutable_snapshot"]) + finally: + store.close() + + def result_documents(self, run_id): + result = self.service.result(run_id) + artifacts = {item["name"]: item for item in result["artifacts"]} + self.assertTrue(TERMINAL_REPORT_NAMES <= set(artifacts)) + receipt = json.loads(Path(artifacts["receipt.json"]["path"]).read_text()) + return receipt, artifacts + + def finish_valid_delivery(self, requested_model): + self.configure(requested_model=requested_model) + run_id = self.start(self.native_result(), f"valid-{requested_model}") + self.assertEqual(self.wait(run_id, candidate=True)["state"], "queued") + snapshot = self.snapshot(run_id) + attempt = snapshot["delivery_iterations"][0]["implementation"]["attempt"] + self.assertEqual(attempt["selected_profile"]["profile"]["model_id"], requested_model) + self.assertEqual(attempt["selected_profile"]["profile"]["effort"]["value"], "high") + self.assertEqual(attempt["observed_identity"]["model_id"], self.MODEL) + self.assertIsNone(attempt["observed_identity"]["effort"]) + self.assertIsNone(attempt["observed_identity"]["backing_revision"]) + self.assertEqual(attempt["usage"]["total_tokens"], 19) + self.assertEqual(attempt["native_ids"]["session_id"], "native-delivery-session") + self.assertTrue(self.service.resume(run_id)["launched"]) + status = self.wait(run_id) + self.assertEqual(status["state"], "awaiting_host", status) + claimed = self.service.handoff_claim(run_id, status["version"], "identity-host") + packet = claimed["handoff"]["packet"] + completed = self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + packet, "identity-accept", "accept", "Independent native review passed.", + ), + ) + self.assertEqual(completed["state"], "succeeded") + receipt, artifacts = self.result_documents(run_id) + self.assertEqual([item["role"] for item in receipt["attempts"]], ["implementer", "reviewer"]) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertIn("candidate-1.patch", artifacts) + self.fixture.assert_source_unchanged() + + def test_exact_native_identity_completes_public_delivery(self): + self.finish_valid_delivery(self.MODEL) + + def test_alias_resolves_to_observed_identity_through_public_delivery(self): + self.finish_valid_delivery("sonnet") + + def test_invalid_identity_never_publishes_candidate_and_retains_failed_evidence(self): + self.configure() + documents = {} + documents["missing"] = self.native_result() + del documents["missing"]["modelUsage"] + documents["multiple"] = self.native_result() + documents["multiple"]["modelUsage"]["claude-haiku-4-5-20251001"] = self.counters() + documents["mismatch"] = self.native_result() + documents["mismatch"]["modelUsage"] = {"claude-opus-4-6": self.counters()} + for kind, document in documents.items(): + with self.subTest(kind=kind): + run_id = self.start(document, f"invalid-{kind}") + status = self.wait(run_id, candidate=True) + self.assertEqual(status["state"], "failed", status) + self.assertIsNone(status["handoff"]) + receipt, artifacts = self.result_documents(run_id) + self.assertNotIn("candidate-1.json", artifacts) + self.assertNotIn("candidate-1.patch", artifacts) + self.assertEqual(receipt["state"], "failed") + self.assertEqual(receipt["accounting"]["worker_invocations"], 1) + self.assertEqual(len(receipt["attempts"]), 1) + failed_attempt = receipt["attempts"][0] + self.assertEqual(failed_attempt["role"], "implementer") + diagnostics = failed_attempt["native_diagnostics"] + self.assertEqual(diagnostics["schema_version"], 1) + self.assertEqual(diagnostics["identity_status"], "unverified") + self.assertTrue(diagnostics["reason"]) + self.assertEqual(diagnostics["usage"]["total_tokens"], 19) + self.assertEqual(diagnostics["session_id"], "native-delivery-session") + self.assertEqual( + diagnostics["output_sha256"], + hashlib.sha256(self.output.read_bytes()).hexdigest(), + ) + self.assertEqual(diagnostics["output_bytes"], self.output.stat().st_size) + # Bound native diagnostics must survive in the public receipt, + # not only in an unreferenced private stderr traceback. + for reported_model in document.get("modelUsage", {}): + self.assertIn(reported_model, diagnostics["model_usage"]) + captures = [name for name in artifacts if name.endswith((".stdout", ".stderr"))] + self.assertEqual(len(captures), 2) + reopened = Service(self.fixture.runtime) + self.assertEqual(reopened.result(run_id), self.service.result(run_id)) + replayed = reopened.start(self.task(), f"invalid-{kind}") + self.assertFalse(replayed["created"]) + self.assertEqual(replayed["run_id"], run_id) + self.assertEqual(replayed["state"], "failed") + store = Store(reopened.database, reopened.artifacts) + try: + self.assertEqual(store.worker_invocations(run_id), 1) + finally: + store.close() + self.fixture.assert_source_unchanged() + + def test_same_reported_model_cannot_pass_host_independence_gate(self): + self.assert_independence_blocked("host") + + def test_same_reported_model_cannot_pass_headless_independence_gate(self): + self.assert_independence_blocked("headless") + + def fallback_output(self, *, delay=False): + """Install bounded fake native outputs selected by frozen profile model.""" + fallback_result = self.native_result() + fallback_result["modelUsage"] = {self.FALLBACK_MODEL: self.counters()} + fallback_result["session_id"] = "native-fallback-session" + output_path = self.root / "fallback-result.json" + output_path.write_text(json.dumps(fallback_result)) + self.fallback_marker = self.root / "fallback-entered" + pause = "/bin/sleep 5\n" if delay else "" + self.fake_claude.write_text( + "#!/bin/sh\n" + "if [ \"$1\" = \"--version\" ]; then\n" + " printf '%s\\n' '2.1.220 (Claude Code)'\n" + " exit 0\n" + "fi\n" + "previous=''\n" + "for argument in \"$@\"; do\n" + f" if [ \"$previous\" = '--model' ] && [ \"$argument\" = '{self.FALLBACK_MODEL}' ]; then\n" + f" printf '%s\\n' 'started' > {shlex.quote(str(self.fallback_marker))}\n" + f" {pause}" + " printf '%s\\n' \"VALUE = 'fixed'\" > src/app.py\n" + f" /bin/cat {shlex.quote(str(output_path))}\n" + " exit 0\n" + " fi\n" + " previous=$argument\n" + "done\n" + f"/bin/cat {shlex.quote(str(self.output))}\n" + ) + failed_result = self.native_result() + failed_result["modelUsage"] = {"claude-haiku-4-5-20251001": self.counters()} + return failed_result + + def assert_failed_native_attempt_preserved(self, receipt): + failed = receipt["attempts"][0] + self.assertEqual(failed["role"], "implementer") + self.assertEqual(failed["status"], "failed") + self.assertEqual(failed["selected_profile"]["profile_id"], "fixture-implementer") + self.assertIsNone(failed["observed_identity"]) + self.assertEqual(failed["usage"]["total_tokens"], 19) + self.assertEqual(failed["usage"]["source"], "native_reported") + diagnostic = failed["native_diagnostics"] + self.assertEqual(diagnostic, failed["error"]["native_diagnostics"]) + self.assertEqual(diagnostic["identity_status"], "unverified") + self.assertEqual(diagnostic["session_id"], "native-delivery-session") + self.assertEqual(diagnostic["usage"], failed["usage"]) + self.assertIn("claude-haiku-4-5-20251001", diagnostic["model_usage"]) + self.assertEqual(diagnostic["output_sha256"], hashlib.sha256(self.output.read_bytes()).hexdigest()) + self.assertEqual(receipt["accounting"]["attempt_usage"][0], failed["usage"]) + + def ready_handoff(self, run_id): + self.assertEqual(self.wait(run_id, candidate=True)["state"], "queued") + self.assertTrue(self.service.resume(run_id)["launched"]) + status = self.wait(run_id) + self.assertEqual(status["state"], "awaiting_host", status) + return self.service.handoff_claim(run_id, status["version"], "identity-host") + + def test_native_fallback_keeps_failed_identity_and_usage_in_success_receipt(self): + self.configure(fallback=True) + run_id = self.start(self.fallback_output(), "native-identity-fallback") + claimed = self.ready_handoff(run_id) + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + claimed["handoff"]["packet"], "accept-native-fallback", "accept", + "Verified fallback received independent native review.", + ), + ) + receipt, _ = self.result_documents(run_id) + self.assertEqual(receipt["state"], "succeeded") + self.assert_failed_native_attempt_preserved(receipt) + implementers = [item for item in receipt["attempts"] if item["role"] == "implementer"] + self.assertEqual(len(implementers), 2) + self.assertEqual(implementers[1]["selected_profile"]["profile_id"], "fixture-implementer-fallback") + self.assertEqual(implementers[1]["observed_identity"]["model_id"], self.FALLBACK_MODEL) + self.assertEqual( + {item["selected_profile"]["profile"]["permission_policy"] for item in implementers}, + {"workspace_write"}, + ) + self.assertEqual(receipt["accounting"]["worker_invocations"], 3) + self.fixture.assert_source_unchanged() + + def test_cancellation_keeps_prior_failed_native_identity_and_usage(self): + self.configure(fallback=True) + run_id = self.start(self.fallback_output(delay=True), "cancel-native-fallback") + deadline = time.monotonic() + 15 + while time.monotonic() < deadline and not self.fallback_marker.exists(): + self.assertNotIn(self.service.status(run_id)["state"], {"failed", "cancelled", "succeeded"}) + time.sleep(0.05) + self.assertTrue(self.fallback_marker.exists(), "fallback never reached its bounded native call") + self.service.cancel(run_id) + self.assertEqual(self.wait(run_id)["state"], "cancelled") + receipt, _ = self.result_documents(run_id) + self.assert_failed_native_attempt_preserved(receipt) + self.assertEqual(len(receipt["attempts"]), 2) + self.assertEqual(receipt["attempts"][1]["status"], "cancelled") + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + self.assertEqual(receipt["lead"]["status"], "not_reached") + self.fixture.assert_source_unchanged() + + def downgrade_saved_identity_for_legacy_fixture(self, run_id): + """Simulate pre-v2 persisted identity; never use as qualification proof.""" + store = Store(self.service.database, self.service.artifacts) + try: + snapshot = json.loads(store.run(run_id)["mutable_snapshot"]) + implementation = snapshot["delivery_iterations"][0]["implementation"] + implementation["schema_version"] = 1 + attempt = implementation["attempt"] + old_fields = { + "harness", "harness_version", "model_provider", "model_id", + "effort", "permission_policy", "verification", + } + attempt["observed_identity"] = { + key: value for key, value in attempt["observed_identity"].items() + if key in old_fields + } + attempt["observed_identity"]["effort"] = attempt["selected_profile"]["profile"]["effort"]["value"] + store.connection.execute( + "UPDATE runs SET mutable_snapshot=? WHERE id=?", (canonical_json(snapshot), run_id), + ) + finally: + store.close() + + def test_legacy_identity_cannot_authorize_new_acceptance_but_rejection_replays(self): + self.configure() + run_id = self.start(self.native_result(), "legacy-new-acceptance") + claimed = self.ready_handoff(run_id) + packet = claimed["handoff"]["packet"] + self.downgrade_saved_identity_for_legacy_fixture(run_id) + self.service = Service(self.fixture.runtime) + with self.assertRaisesRegex(ContractError, "requires v2"): + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + packet, "legacy-new-accept", "accept", "Try old copied identity evidence.", + ), + ) + decision = self.fixture.decision(packet, "legacy-reject", "reject", "New verification required.") + self.service.handoff_complete(run_id, claimed["claim"], decision) + receipt, _ = self.result_documents(run_id) + self.assertEqual(receipt["state"], "failed") + self.assertEqual(receipt["delivery_iterations"][0]["implementation"]["schema_version"], 1) + replay = Service(self.fixture.runtime).handoff_complete(run_id, claimed["claim"], decision) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay["state"], "failed") + self.assertEqual(self.result_documents(run_id)[0], receipt) + self.fixture.assert_source_unchanged() + + def test_historical_successful_legacy_receipt_reads_and_exact_acceptance_replays(self): + self.configure() + run_id = self.start(self.native_result(), "historical-legacy-acceptance") + claimed = self.ready_handoff(run_id) + self.downgrade_saved_identity_for_legacy_fixture(run_id) + decision = self.fixture.decision( + claimed["handoff"]["packet"], "historical-accept", "accept", "Historical fixture acceptance.", + ) + # Construct the historical fixture under pre-v2 semantics, then remove + # the bypass before any assertions. This is compatibility data only. + with patch("devsquad.workflows.require_independent_delivery_review", return_value=None): + self.service.handoff_complete(run_id, claimed["claim"], decision) + self.service = Service(self.fixture.runtime) + receipt, _ = self.result_documents(run_id) + self.assertEqual(receipt["state"], "succeeded") + self.assertEqual(receipt["delivery_iterations"][0]["implementation"]["schema_version"], 1) + replay = self.service.handoff_complete(run_id, claimed["claim"], decision) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay["state"], "succeeded") + self.assertEqual(self.result_documents(run_id)[0], receipt) + with self.assertRaises(ContractError): + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + claimed["handoff"]["packet"], "different-new-accept", "accept", "Not an exact replay.", + ), + ) + self.fixture.assert_source_unchanged() + + def assert_independence_blocked(self, mode): + self.configure(requested_model="sonnet", reviewer_model=self.MODEL) + run_id = self.start(self.native_result(), f"same-model-{mode}", mode=mode) + status = self.wait(run_id, candidate=True) + if status["state"] == "queued": + resumed = self.service.resume(run_id) + self.assertTrue(resumed["launched"], resumed) + status = self.wait(run_id) + if status["state"] == "awaiting_host": + claimed = self.service.handoff_claim(run_id, status["version"], "same-model-host") + packet = claimed["handoff"]["packet"] + with self.assertRaises(ContractError): + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + packet, "same-model-accept", "accept", "Try the same model despite new labels.", + ), + ) + self.service.handoff_complete( + run_id, claimed["claim"], self.fixture.decision( + packet, "same-model-reject", "reject", "Independent review was not established.", + ), + ) + status = self.service.status(run_id) + self.assertEqual(status["state"], "failed", status) + receipt, _ = self.result_documents(run_id) + self.assertNotEqual(receipt["lead"].get("disposition"), "accept") + self.assertIn("independen", json.dumps(receipt).lower()) + self.fixture.assert_source_unchanged() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py index 3a1aa8f..6233c44 100644 --- a/test/core/test_delivery_workflow.py +++ b/test/core/test_delivery_workflow.py @@ -384,9 +384,10 @@ def test_frozen_claude_worker_edits_only_the_delivery_workspace(self): " exit 0\n" "fi\n" "printf '%s\\n' \"VALUE = 'fixed'\" > src/app.py\n" - "printf '%s\\n' '{\"type\":\"result\",\"is_error\":false," + "printf '%s\\n' '{\"type\":\"result\",\"subtype\":\"success\",\"is_error\":false," "\"result\":\"Applied the bounded fix.\"," "\"session_id\":\"session-fixture\"," + "\"modelUsage\":{\"claude-sonnet-fixture\":{\"inputTokens\":12,\"outputTokens\":7}}," "\"usage\":{\"input_tokens\":12,\"output_tokens\":7}}'\n" ) binary.chmod(0o700) diff --git a/test/core/test_implementation_identity_evidence.py b/test/core/test_implementation_identity_evidence.py new file mode 100644 index 0000000..4840864 --- /dev/null +++ b/test/core/test_implementation_identity_evidence.py @@ -0,0 +1,225 @@ +"""Adversarial native evidence parsing and coordinator import validation.""" + +from __future__ import annotations + +import copy +import hashlib +import json +import unittest +from unittest.mock import patch + +import test_claude_identity as fixtures +from devsquad.claude_delivery_worker import run +from devsquad.claude_identity import ( + ClaudeResultError, failure_envelope, strict_json, validate_failure, +) +from devsquad.contracts import ContractError +from devsquad.supervisor import Supervisor +from devsquad.workflows import validate_implementation_evidence + + +class ImplementationIdentityEvidenceTest(unittest.TestCase): + def setUp(self): + self.fixture = fixtures.ClaudeImplementationIdentityTest(methodName="runTest") + self.fixture.setUp() + self.addCleanup(self.fixture.doCleanups) + + def evidence(self): + self.fixture.output.write_text(json.dumps(self.fixture.document())) + snapshot = self.fixture.snapshot() + return snapshot, run(snapshot) + + def test_strict_native_json_rejects_duplicates_and_nonfinite_numbers(self): + document = json.dumps(self.fixture.document()) + for payload in ( + document[:-1] + ', "session_id": "other"}', + document[:-1] + ', "unrelated": NaN}', + document[:-1] + ', "unrelated": Infinity}', + document[:-1] + ', "unrelated": 1e9999}', + ): + with self.subTest(payload=payload[-80:]): + self.fixture.output.write_text(payload) + with self.assertRaises(ClaudeResultError): + run(self.fixture.snapshot()) + + def test_optional_native_usage_metadata_is_typed_and_bounded(self): + for field, invalid in ( + ("costUSD", -1), ("costUSD", True), ("costUSD", 10 ** 400), + ("cacheReadInputTokens", -1), ("contextWindow", True), + ("canonicalModel", {}), ("provider", "other-provider"), + ("inputTokens", 2 ** 63), + ): + with self.subTest(field=field, invalid_type=type(invalid).__name__): + document = self.fixture.document() + document["modelUsage"][self.fixture.MODEL][field] = invalid + with self.assertRaises(ContractError): + self.fixture.run_document(document) + + def test_missing_aggregate_usage_stays_unknown_with_valid_identity(self): + document = self.fixture.document() + del document["usage"] + attempt = self.fixture.run_document(document)["attempt"] + self.assertEqual(attempt["usage"]["source"], "unavailable") + self.assertIsNone(attempt["usage"]["total_tokens"]) + self.assertEqual(attempt["observed_identity"]["verification_scope"], "reported_model") + self.assertIsNone(attempt["native_model_requests"]) + + def test_coordinator_rejects_tampering_before_candidate_finalization(self): + snapshot, evidence = self.evidence() + mutations = [ + (("schema_version",), 1), + (("attempt", "observed_identity"), {}), + (("attempt", "observed_identity", "model_id"), "claude-other"), + (("attempt", "observed_identity", "effort"), "high"), + (("attempt", "observed_identity", "backing_revision"), "invented"), + (("attempt", "observed_identity", "verification_scope"), "everything"), + (("attempt", "observed_identity", "model_source"), "requested_argv"), + (("attempt", "observed_identity", "harness_version"), "different"), + (("attempt", "observed_identity", "native_evidence"), None), + (("attempt", "observed_identity", "native_evidence", "schema_version"), True), + (("attempt", "observed_identity", "native_evidence", "is_error"), 0), + (("attempt", "observed_identity", "native_evidence", "model_usage"), {}), + (("attempt", "native_ids", "session_id"), "different-session"), + (("attempt", "usage", "input_tokens"), 999), + ] + for path, value in mutations: + with self.subTest(path=path): + tampered = copy.deepcopy(evidence) + target = tampered + for field in path[:-1]: + target = target[field] + target[path[-1]] = value + with self.assertRaises(ContractError): + validate_implementation_evidence(tampered, snapshot) + with patch("devsquad.supervisor.freeze_delivery_candidate") as freeze: + with self.assertRaises(ContractError): + # This import boundary validates before accessing its store. + Supervisor._commit_delivery_candidate( + object(), "unused", {}, [], {}, snapshot, + json.dumps(tampered).encode(), + ) + freeze.assert_not_called() + + def test_native_claim_cannot_downgrade_to_fixture_evidence(self): + snapshot, evidence = self.evidence() + evidence["schema_version"] = 1 + evidence["attempt"]["observed_identity"] = None + evidence["attempt"]["native_ids"] = {} + with self.assertRaises(ContractError): + validate_implementation_evidence(evidence, snapshot) + + def test_frozen_fallback_is_validated_against_its_own_profile(self): + snapshot, _ = self.evidence() + route = snapshot["routing"]["roles"]["implementer"] + fallback = copy.deepcopy(route["selected"]) + fallback["profile"]["id"] = fallback["profile_id"] = "fallback-writer" + fallback["reference"]["id"] = "fallback-writer" + fallback["profile"]["model_id"] = "sonnet" + fallback["profile_sha256"] = hashlib.sha256(json.dumps( + fallback["profile"], sort_keys=True, separators=(",", ":"), + ).encode()).hexdigest() + route["fallbacks"] = [fallback] + snapshot["implementation_adapters"]["fallback-writer"] = snapshot["implementation_adapter"] + launched = copy.deepcopy(snapshot) + launched["routing"]["roles"]["implementer"]["selected"] = fallback + evidence = run(launched) + self.assertEqual(validate_implementation_evidence(evidence, snapshot), evidence) + self.assertEqual(evidence["attempt"]["selected_profile"]["profile_id"], "fallback-writer") + + def test_failed_identity_keeps_only_typed_native_diagnostics(self): + document = self.fixture.document() + document["result"] = "private provider prose must not enter diagnostics" + document["modelUsage"]["claude-other-model"] = { + **self.fixture.model_usage(), "unrelated_secret": "do not retain", + } + with self.assertRaises(ClaudeResultError) as caught: + self.fixture.run_document(document) + diagnostics = caught.exception.diagnostics + self.assertEqual(len(diagnostics["model_usage"]), 2) + self.assertEqual(diagnostics["usage"]["total_tokens"], 19) + self.assertNotIn("private provider prose", json.dumps(diagnostics)) + self.assertNotIn("unrelated_secret", json.dumps(diagnostics)) + self.assertNotIn("do not retain", json.dumps(diagnostics)) + value = failure_envelope(caught.exception, "a" * 64) + self.assertEqual(validate_failure(value, "a" * 64), value) + with self.assertRaises(ContractError): + validate_failure(value, "b" * 64) + for field, invalid in (("identity_status", "verified"), + ("output_sha256", "bad"), ("output_bytes", True)): + altered = copy.deepcopy(value) + altered["native_diagnostics"][field] = invalid + with self.assertRaises(ContractError): + validate_failure(altered, "a" * 64) + + def test_import_decoder_rejects_duplicate_native_evidence_keys(self): + snapshot, evidence = self.evidence() + encoded = json.dumps(evidence) + encoded = encoded[:-1] + ', "schema_version": 2}' + with self.assertRaises(ContractError): + strict_json(encoded) + with patch("devsquad.supervisor.freeze_delivery_candidate") as freeze: + with self.assertRaises(ContractError): + Supervisor._commit_delivery_candidate( + object(), "unused", {}, [], {}, snapshot, encoded.encode(), + ) + freeze.assert_not_called() + + def test_import_must_match_actual_attempt_not_another_allowed_fallback(self): + snapshot, evidence = self.evidence() + route = snapshot["routing"]["roles"]["implementer"] + primary = route["selected"] + fallback = copy.deepcopy(primary) + fallback["profile"]["id"] = fallback["profile_id"] = "fallback-writer" + fallback["reference"]["id"] = "fallback-writer" + fallback["profile_sha256"] = hashlib.sha256(json.dumps( + fallback["profile"], sort_keys=True, separators=(",", ":"), + ).encode()).hexdigest() + route["fallbacks"] = [fallback] + snapshot["implementation_adapters"]["fallback-writer"] = snapshot["implementation_adapter"] + evidence["attempt"]["selected_profile"] = fallback + # It is valid evidence for this frozen fallback, but not for the actual + # primary invocation importing it. Validate the coordinator boundary. + validate_implementation_evidence(evidence, snapshot) + for attempt in ( + {"profile_index": 0, "profile_id": primary["profile_id"]}, + {"profile_index": 0, "profile_id": fallback["profile_id"]}, + {"profile_index": True, "profile_id": fallback["profile_id"]}, + ): + with self.subTest(attempt=attempt): + with patch("devsquad.supervisor.freeze_delivery_candidate", side_effect=AssertionError("unbound evidence reached freeze")) as freeze: + with self.assertRaises(ContractError): + Supervisor._commit_delivery_candidate( + object(), "unused", attempt, [], {}, snapshot, + json.dumps(evidence).encode(), + ) + freeze.assert_not_called() + + def test_invalid_utf8_stdout_retains_original_native_byte_hash(self): + payload = b"\xff\xfe\r\n" + self.fixture.output.write_bytes(payload) + with self.assertRaises(ClaudeResultError) as caught: + run(self.fixture.snapshot()) + diagnostics = caught.exception.diagnostics + self.assertEqual(diagnostics["output_bytes"], len(payload)) + self.assertEqual(diagnostics["output_sha256"], hashlib.sha256(payload).hexdigest()) + self.assertIsNone(diagnostics["session_id"]) + + def test_failed_result_hash_preserves_crlf_native_bytes(self): + document = self.fixture.document() + document["modelUsage"]["claude-other-model"] = self.fixture.model_usage() + payload = (json.dumps(document, indent=2) + "\n").replace("\n", "\r\n").encode() + self.fixture.output.write_bytes(payload) + with self.assertRaises(ClaudeResultError) as caught: + run(self.fixture.snapshot()) + self.assertEqual(caught.exception.diagnostics["output_bytes"], len(payload)) + self.assertEqual(caught.exception.diagnostics["output_sha256"], hashlib.sha256(payload).hexdigest()) + + def test_invalid_utf8_stderr_cannot_preempt_valid_result_parsing(self): + content = self.fixture.binary.read_text() + self.fixture.binary.write_text(content.replace("/bin/cat", "printf '\\377' >&2\n/bin/cat")) + evidence = self.fixture.run_document(self.fixture.document()) + self.assertEqual(evidence["attempt"]["observed_identity"]["model_id"], self.fixture.MODEL) + + +if __name__ == "__main__": + unittest.main() From b3de5c6049a557c2c75d2a48754ec45cba4f898f Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 00:14:53 -0700 Subject: [PATCH 135/197] WIP checkpoint: docs: refresh SOL execution plan and record R2 offline verification (2026-10-01 00:14) --- docs/plans/engineering-team/M5-STATUS.md | 22 +-- docs/plans/engineering-team/RESUME.md | 39 +++-- .../engineering-team/SOL-REVIEW-FOLLOWUP.md | 155 +++++++++++++----- docs/plans/engineering-team/backlog.json | 9 +- .../R2-observed-identity-2026-10-01.json | 88 ++++++++++ 5 files changed, 243 insertions(+), 70 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R2-observed-identity-2026-10-01.json diff --git a/docs/plans/engineering-team/M5-STATUS.md b/docs/plans/engineering-team/M5-STATUS.md index 6aea58b..e0ab10d 100644 --- a/docs/plans/engineering-team/M5-STATUS.md +++ b/docs/plans/engineering-team/M5-STATUS.md @@ -3,26 +3,28 @@ M5 is **in progress with independent repairs available**. The September 29 review at `f4fa657` reopened candidate integrity (F1), observed Claude identity (F2) and complete normal-command check coverage (G4). F1/R1 is now repaired -and verified offline; execute R2/R6 in +and verified offline; R2 is also source/offline verified at `ef98889`. Execute R6 in [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The genuine Claude-to-Codex live gate additionally remains blocked on normal Claude login. -R2 now has a preserved red baseline: 16 fake-CLI worker tests ran with 39 -assertion/subtest failures and 4 errors against unchanged production source -at `6848f11`. See [baseline evidence](evidence/R2-identity-red-baseline-2026-09-29.json). -Parser repair, strict imported evidence, public independent-review acceptance -and durable failure/replay regressions remain unfinished. This checkpoint -does not establish a green core suite or a corrected Claude live receipt. +R2 now enforces native reported model identity, strict imported evidence bound +to the actual attempt, independent-review acceptance and durable failure/ +fallback/cancellation/legacy replay. Its 37 added regressions are included in +the final 367-test suite (OK, two optional-SDK skips); 227 Bash assertions and +the generated-reference check passed. The SQLite cleanup warning remains R5. +See [R2 evidence](evidence/R2-observed-identity-2026-10-01.json); the +[red baseline](evidence/R2-identity-red-baseline-2026-09-29.json) is retained as +history. This is not a corrected installed Claude live receipt or full M5 closure. | Requirement | Planned evidence | Status | |---|---|---| -| Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | argv conformance verified at `d96e9e4`; observed identity reopened under R2 | +| Claude headless adapter | Manifest/argv conformance, exact model and effort validation, structured result faults, bounded permission/tool surface, recursion guard and installed-wheel contents | argv conformance verified at `d96e9e4`; R2 observed-model evidence verified offline; unreported effort remains unknown | | Isolated implementation | Run-owned detached delivery worktree at the frozen target, one active writer and original checkout/index/HEAD preservation | verified offline at `0e88d73` | | Scoped local candidate | Out-of-scope and symlink-escape rejection; intentional untracked capture; local candidate commit and patch/hash artifacts; no merge, push or remote mutation | verified offline at `0e88d73` | -| Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | reopened under R2; prior fixtures do not detect fabricated observed identity; live proof pending | +| Independent reviewer | Different verified model identity is mandatory and a different harness is preferred when qualified; unknown/same identity cannot count | R2 public host/headless identity gates verified offline; installed/live proof pending | | Candidate-bound review/checks | Read-only review and separate check worktree bind to the exact candidate; changed candidate invalidates prior evidence | R1 check-mutation repair verified offline; R6 normal-command check coverage remains | | Bounded correction/fallback | Seeded defect causes revise to implementation, then new review/checks; rate-limit fallback retains permissions and all finite budgets | correction/budgets verified at `4c76887`; same-permission delivery fallback verified at `b7d90cc` | -| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | R1 candidate integrity now enforced; R2 identity repair remains | +| Non-overridable disposition | Missing implementation/invalid review/mandatory failing check block acceptance regardless of lead prose | R1 candidate integrity and R2 actual-model identity enforced offline; installed/live proof pending | | Complete result history | Receipt retains every implementer/reviewer/lead attempt, failed fallback, repair, revision, candidate and evidence hash | success, repair, fallback, failure and cancellation history verified offline at `b7d90cc` | | Crash recovery | Killing a live implementation supervisor cannot create a duplicate writer on resume | prelaunch and live revised-writer recovery verified offline at `f199cd2` | | Live acceptance | One bounded issue completes across at least two authenticated subscription harnesses with different verified models | pending repairs and normal Claude CLI login | diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 615c109..114af48 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,7 +6,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Review correction and next action -R2 implementation is preserved after `3612942`: strict native result parsing, +R2 source/offline repair is verified at `ef98889`: strict native result parsing, v2 import evidence, actual-model independence and durable failed diagnostics. The 16 original worker regressions, 12 import/parser regressions, 19 existing delivery tests and nine public native delivery/fallback/cancellation/legacy @@ -21,15 +21,22 @@ failures and two skips**, not a pass. Four failures show budget expiration or 15–17 minute UTC jumps during a 231-second monotonic suite. On October 1, all six failed cases passed unchanged in 18.998 seconds; per-test UTC and monotonic elapsed measurements agreed. This supports an environmental timing explanation, -but does not erase the failed gate. A clean complete rerun remains the next -verification step; do not weaken budget or identity enforcement. The recurring -SQLite cleanup warning remains assigned to R5. **R2 is not closed yet.** +but does not erase the failed gate. The final unchanged-source rerun completed +**367 tests in 215.886 seconds, OK with two optional-SDK skips**; its wrapper +measured 216.021 UTC seconds and 216.020 monotonic seconds. The SQLite cleanup +warning still appeared and remains assigned to R5. All 227 Bash assertions +and the generated-reference check passed. **R2 source/offline closure is +recorded; installed/live closure remains R8.** See +[R2 evidence](evidence/R2-observed-identity-2026-10-01.json). No gate is running. No live provider call, installed refresh or global provider-setting change occurred. The original red baseline is retained in [R2 baseline evidence](evidence/R2-identity-red-baseline-2026-09-29.json). -Once R2's final gate is green and recorded, continue R3 rather than retrying -blocked authentication. R3's two-outcome/three-case promotion bug was reproduced +The latest user request was to create a plan and feedback for SOL. This +checkpoint updates the canonical handoff and verification records only; no +new implementation was added. SOL should begin **R3a** in +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), not redo R2 or retry blocked +authentication. R3's two-outcome/three-case promotion bug was reproduced again without modifying production code; it still needs provenance repair. R1's source repair is verified after `399d93d`: the public regression first @@ -49,16 +56,16 @@ is verified with installed refresh pending; M4 retains its real-Claude-host gate independent repairs and integration work; C1 remains pending full-delivery scope. R1 is the first implemented repair; do not confuse it with full closure. -Execute [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), starting at **R2**: -repair Claude identity using native `modelUsage`, keeping effective effort -unknown when unreported. The plan contains the bounded local/primary-source -investigation; no live Claude generation was used. Continue with -independent experiment evidence (R3), routing/catalog/quota (R4), learning +Execute [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), starting at **R3**: +bind independent experiment evidence to saved runs, frozen assignments, +profiles and paired inputs; protect replay and future qualification decisions +against stale or unverified evidence. The plan contains bounded source-review +feedback and acceptance tests. Continue with routing/catalog/quota (R4), learning runtime connections (R5), and normal terminal/readiness/check discovery (R6). R7 covers C1 and R8 covers installed/live closure. Authentication and the Jev key block their specific live subgates, not the independent engineering work. -Fresh review verification ran 317 Python tests successfully (2 optional-SDK +The original September 29 review ran 317 Python tests successfully (2 optional-SDK skips), 227 Bash assertions, generated-reference validation and an installed payload comparison. An unclosed SQLite `ResourceWarning` still appeared and is assigned to R5. Existing live receipts remain evidence of their exact runs, @@ -340,10 +347,10 @@ provider paths must not be advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Implement R2 against the saved 16-test failing Claude-identity baseline; - add strict import, actual-model independence and durable failure/replay - public regressions. Then execute R3 and R4–R6 in dependency order. Preserve - the verified R1 repair, +2. Begin R3a: add the two-outcome/three-case regression and strict frozen + experiment provenance contract. Finish R3b/R3c saved evidence, replay, + qualification/promotion/rollback/catalog-fallback eligibility and historical + compatibility, then execute R4–R6 in dependency order. Preserve R1/R2, explicit check `output_paths` contract and historical receipts. Do not rewrite the architecture or reset completed work. 3. Keep Claude and Grok live probes paused until normal login is restored. diff --git a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md index a9b5160..9c96614 100644 --- a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md +++ b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md @@ -1,4 +1,4 @@ -# Sol execution plan and review feedback — September 29, 2026 +# Sol execution plan and review feedback — October 1, 2026 ## Assignment and starting point @@ -20,29 +20,38 @@ Read `AGENTS.md`, `CONTRIBUTING.md`, [RESUME.md](RESUME.md), this document and for each work package. Do not reload the historical chat or restart M1/M2. This review supersedes the earlier claim that only credentials remain. -Current execution checkpoint: **R1 source repair verified**, with public -mutation regressions, a ten-case workspace mutation matrix, historical replay -coverage and 330-test core gate (two optional-SDK skips). See -[R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json). Start at **R2** -after checking current Git state. R2–R8 and installed refresh remain open. +Current execution checkpoint: **R1 and R2 source/offline repairs verified**. +The implementation baseline is `ef988897e2dda495482dd26e7c583bc428086ba0`; +preserve any later commits. R2's final complete run passed 367 tests with two +optional-SDK skips in 215.886 seconds. The SQLite finalizer warning remains +open under R5; this is not a warning-free or installed/live pass. See +[R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json) and +[R2 evidence](evidence/R2-observed-identity-2026-10-01.json). Start at **R3**. +R3–R8, C1 and affected installed/live proofs remain open. ### Ready-to-execute handoff for Sol -Production baseline is now `6848f1156bee10e1ef2617402dfc7eaebdc58076` (R1). -R2 has a saved **red test checkpoint**, not an implemented repair: -`test/core/test_claude_identity.py` contains 16 offline fake-CLI tests. The -latest focused run completed in 3.162 seconds with **39 assertion/subtest -failures and 4 errors**. The errors expose non-object JSON reaching an unsafe -`.get()` call. See [R2 baseline evidence](evidence/R2-identity-red-baseline-2026-09-29.json). -No production source changed for this planning handoff. The prior 330-test -green gate belongs to R1; the current tree includes known failing R2 tests. +Do not reimplement R2. It now has a shared strict native parser, v2 imported +evidence, actual-attempt profile binding, actual-model independence and durable +failure/fallback/cancellation/history handling. The original 16 worker tests, +12 import/parser tests and nine public identity tests cover this repair. +Two independent-review findings were reproduced and repaired: another allowed +fallback profile cannot impersonate the actual attempt, and invalid UTF-8 or +CRLF native output retains its exact byte hash. The original +[red baseline](evidence/R2-identity-red-baseline-2026-09-29.json) remains history. + +Preserve the verification history: a prior 367-test run failed six cases; +all six passed unchanged with stable clocks, and the final full run passed +with UTC and monotonic elapsed times agreeing. Do not erase the failed run or +weaken budgets to make tests pass. The current request is a planning handoff: +no additional implementation, installation or provider call was performed. | Next slice | Deliverable | Gate before claiming completion | |---|---|---| -| R2a | Native identity parser and worker evidence | Existing 16 tests green; explicit provenance and unknown settings | -| R2b | Strict import plus actual-model independence | Tampered evidence and same effective model rejected through public host/headless paths | -| R2c | Durable identity failures and replay compatibility | Invalid identity retained in failed receipts; historical reads work without authorizing new acceptance | -| R3 | Independent trial evidence | No reused observation can inflate evaluation or held-out sample size | +| R1 / R2 | Preserve verified source repairs | Recheck affected regressions when shared code changes; installed/live proof remains R8 | +| R3a | Strict experiment provenance contract | Reused outcomes/runs/splits and mismatched paired inputs rejected | +| R3b | Durable validation and current-evidence eligibility | Replay, qualification, promotion, rollback and catalog fallback cannot reuse stale/unverified evidence | +| R3c | Public regression and migration evidence | Valid independent pairs work; legacy receipts remain readable; late corrections block new unsafe decisions | | R4 → R5 → R6 | Routing, learning and normal terminal integration | Public-run evidence through the full chain, then fresh-install usability proof | | R7 / R8 | Council and installed/live closure | Each separately required acceptance gate; external blockers remain explicit | @@ -69,7 +78,7 @@ Passing counts do not demonstrate that the complete workflow enforces its contract. Each new completion claim needs an actual runtime caller, a public path regression and the corresponding installed/live proof where required. -During this review, the complete Python suite ran 317 tests successfully +During the original September 29 review, the complete Python suite ran 317 tests successfully (two optional-SDK skips), and all 227 Bash assertions passed. The generated reference and installed-source comparison passed. A recurring unclosed SQLite `ResourceWarning` appeared despite process exit zero; it remains an @@ -101,7 +110,7 @@ claim routing improvement from a synthetic smoke result. ## Work order and ownership -R1, R2 and R3 come first. R4 precedes R5; R6 integrates the repaired public +R1 and R2 source repairs are done; R3 is the next dependency. R4 precedes R5; R6 integrates the repaired public paths. R7 follows the relevant M3–M6 repairs. R8 proves the installed product; its authentication and classifier subgates may remain externally blocked. Continue all independent work when a live subgate is blocked. @@ -163,7 +172,10 @@ model, absent identity, multiple models, unavailable effort and tampered evidence all have explicit tested results. The genuine two-harness live gate remains open until the corrected adapter obtains a real receipt. -#### R2 implementation feedback and missing tests +#### R2 implemented contract to preserve + +The following requirements are implemented at `ef98889`, not a new to-do +list. Keep them enforced when R3–R6 touch shared import/disposition paths. - **R2a — parser/worker:** define a shared strict identity contract rather than letting the worker and importer invent separate interpretations. Require a @@ -179,8 +191,8 @@ remains open until the corrected adapter obtains a real receipt. `model` is not identity authority either, but a contradictory serving-model claim fails closed. Keep effective effort/backing revision null when unreported, and make the scope of verification explicit. The new - test file covers this layer only; update older success fixtures to include - valid native evidence instead of weakening the parser. + worker test file covers this layer only; import and public runtime suites + provide the additional coverage. Fixtures must contain valid native evidence. - **R2b — import/acceptance:** validate the native evidence against the frozen adapter, selected profile (including a legitimate fallback), session and claimed observation. A dictionary or `verification="verified"` label is not @@ -199,18 +211,17 @@ remains open until the corrected adapter obtains a real receipt. new acceptance. Version evidence/contracts explicitly if needed. Preserve R1's non-overridable check-integrity gates throughout. -Reproduce the saved baseline with: +Run the preserved worker regression set with: ```bash PYTHONDONTWRITEBYTECODE=1 PYTHONPATH=plugin/core/src \ python3 -m unittest discover -s test/core -p test_claude_identity.py -v ``` -This command is expected to fail until R2 is implemented. Do not skip or mark -the regressions expected-failure merely to recover a green suite. Before R2 -offline closure, run its added public tests, complete core discovery, the Bash -suite and generated-reference check, then obtain one bounded independent -patch review. Only after that consider the separately blocked live receipt. +This command is now expected to pass. Its historical red result belongs to +`6848f11` and the saved baseline artifact. The added public tests, full core +gate and independent patch review have completed; do not substitute these +offline results for the separately blocked installed/live receipt. ### R3 — Make experiment and held-out evidence independent @@ -238,17 +249,67 @@ tasks fail; valid disjoint pairs qualify; failures/missingness remain visible; replay cannot bypass the repaired gate. Promotion and rollback use the same validated evidence path. +#### R3 execution slices and review feedback + +Source inspection at `ef98889` confirms a broader problem than duplicate IDs: +`learning.py:343` accepts profile IDs without fingerprints; +`store.py:1719` loads outcome payloads without run/snapshot/attempt provenance; +evaluation replay (`store.py:1696`) and qualification replay +(`store.py:1994`) return before current evidence validation. Fix the complete +trust path, not only the sample counter. + +1. **R3a — contract and red tests.** Define a versioned frozen assignment: + experiment/spec hash, project, case, split, arm, tested role, concrete + profile hash and paired-input hash. Reject global outcome reuse, underlying + run reuse and evaluation/held-out input overlap. Review arms must share + the candidate; implementation arms must share baseline and normalized + task/scope/acceptance/checks. Explicitly define which incidental paths and + tested routing field are excluded from paired-input hashing; do not drop + substantive inputs to obtain a match. +2. **R3b — saved evidence and shared eligibility.** Join outcomes to saved + runs, project identity, frozen snapshots and actual role attempts. The + assignment/profile must agree with actual execution, including fallbacks. + Persist final-outcome and correction-chain hashes in a versioned evaluation. + One shared current-evidence gate must protect qualification, evaluation/ + qualification replay, new promotion, regression rollback and unavailable- + profile catalog fallback. A new correction invalidates old eligibility; + require an explicit new evaluation/review, not silent receipt rewriting. +3. **R3c — compatibility and public proof.** Preserve historical v1 records + and exact completed-decision replay without another binding mutation. + Label unverified historical evidence ineligible for new decisions. Replace + positive tests that SQL-terminalize empty runs with distinct saved runs, + frozen assignments and bound actual attempts. Keep R5's public trial + controller as a separate deliverable; R3 must not claim it already exists. + +Use this regression order: two outcomes reused across three cases; cross-arm +and underlying-run reuse; wrong project/experiment/case/split; same profile ID +with a different fingerprint; wrong actual fallback; mismatched candidate or +baseline/task/checks; valid disjoint pairs; failed/missing arms; atomic +persistence and replay; legacy read versus new decision; late escaped defect +after evaluation and after qualification; valid promotion/rollback and stale +catalog-fallback rejection. Record independent pairs, not the number of labels. + +R3 can use the existing run IDs, project Git common directory, request hash, +frozen task/config hashes, candidate/baseline identity, selected profiles and +attempt profile indices. Do not trust caller-supplied `experimental` labels +as proof. Failed and missing evidence must remain visible and ineligible for +unearned success credit. Keep schema/contract changes and migrations together. + ### R4 — Connect normal routing, catalog lifecycle and quota **Files:** `task_entry.py`, `catalog.py`, `capacity.py`, `router.py`, `service.py`, `store.py`, adapter metadata and relevant tests. 1. Resolve ordinary roles through approved stable aliases and existing - policy. Preserve explicit concrete pins and qualified fallbacks. Bootstrap + policy. Preserve explicit concrete pins, their pinned-selection provenance + and qualified fallbacks. A normal `--model` or `--review-model` selection + must not be reported as automatic merely because task preparation embedded + it as a singleton profile. Bootstrap without evidence must remain an explicitly labelled bounded trial and cannot become a proven default through catalog ordering or provider hints. 2. Connect discovery to account/config/version-scoped last-good snapshots, - TTL, a refresh lease, bounded timeout/backoff and pagination. Wire complete + TTL, a refresh lease, bounded timeout/backoff and existing protocol pagination. + Reuse `codex_protocol.py` pagination rather than replacing it. Wire complete drift into affected-profile revalidation and qualified fallback/block. Incomplete discovery must preserve the incumbent and prior snapshot. 3. Normalize documented native Codex quota observations into existing typed @@ -262,6 +323,8 @@ partial/auth-failed discovery does not remove models; two projects honor fresh weekly exhaustion despite short-window availability; stale/unsupported quota remains unknown; no permission or billing expansion occurs. Exercise these through normal task entry, not only router/store helper calls. +The existing normal profiles are already labelled `trial`; the defect is +bypassing stable alias/qualification policy, not a missing trial label. ### R5 — Feed learning and trials from actual saved runs @@ -303,7 +366,8 @@ guide and public CLI tests. Use the existing service and single lead authority. run; otherwise show choices. Supply readable output and a real next action. 2. Separate installed/registered/supported/authenticated/operation-verified readiness. Use non-generating auth checks where supported and report - unknown otherwise. Missing auth must not advertise an unavailable workflow + unknown otherwise. An unverified CLI version must not count as supported + adapter readiness. Missing auth must not advertise an unavailable workflow as ready. Keep required logins in the user's normal provider flow. 3. Discover checks from the selected committed target and approved project contracts. For DevSquad, include the Bash suite, relevant Python core suite @@ -384,14 +448,25 @@ permission to claim routing improvement. - Report remaining work as concrete acceptance gates, not invented completion percentages or fixed five-hour-window estimates. Shared quota depends on actual models, context and account usage. +- Keep one short recovery entry with the source revision, exact active test + session/log if any, observed failure, next command and changed files. Resume + from that entry rather than repeatedly loading the full conversation. Do + not launch a second full suite while the first remains live. Distinguish + account quota, tool timeout, runtime budget expiry and host clock movement; + they are not interchangeable explanations for an interruption. ## First action for Sol -Read the recovery files, verify current Git state, and start with the saved -16-test R2 red baseline above. Implement R2a, then add the missing R2b/R2c -public regressions and repairs. Preserve R1's check-output contract and -mutation evidence. Continue R2–R6 without waiting on the Claude login or Jev -key. Retain R7 and R8 in the full scope and continue their independent work -as dependencies become ready. Report each slice as red baseline, verified -offline, installed proof, or externally blocked; do not collapse those states -into a single completion claim. +Read the recovery files and verify current Git state. Preserve `ef98889` and +later work. Begin **R3a** with the two-outcome/three-case failing regression and +the versioned provenance contract above, then finish R3b/R3c before R4. +Preserve R1's check-integrity gate and R2's actual-attempt/native-identity gate. +Continue R3–R6 without waiting on the Claude login or Jev key. Retain R7/R8 +and C1 in the full scope. Report each slice as red baseline, verified offline, +installed proof, or externally blocked; never collapse those into one claim. + +Suggested instruction to SOL: “Execute this plan from R3 on +`codex/engineering-team`, in small spec → failing public test → implementation +→ verification → review → checkpoint cycles. Do not redo R1/R2 or reload the +old chat. After each slice report the evidence, open gates and exact next +action. Keep the classifier off until its separate adoption gate passes.” diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 16f7fda..779cf81 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -23,8 +23,8 @@ }, "review_work_packages": [ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, - {"id": "R2", "title": "Observed Claude execution identity", "status": "in_progress", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-identity-red-baseline-2026-09-29.json", "checkpoint": "Source repair and public identity/failure/fallback/replay regressions implemented; two independent-review findings fixed and retested. Last full gate: 367 tests, six failures, two optional-SDK skips with observed wall-clock budget jumps. All six failures passed unchanged on October 1 with stable UTC/monotonic timing. Final complete gate and evidence checkpoint remain; installed/live proof stays open."}, - {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "pending", "milestones": ["M6"], "depends_on": [], "items": ["F3"]}, + {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, + {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "pending", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "checkpoint": "Next implementation: R3a provenance contract and reused-evidence regression, then R3b shared current-evidence eligibility and R3c public/legacy/correction proof. Source mapping is in SOL-REVIEW-FOLLOWUP.md; no R3 implementation is claimed."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, {"id": "R6", "title": "Normal terminal experience and readiness", "status": "pending", "milestones": ["M5", "M7"], "depends_on": ["R1", "R2", "R4", "R5"], "items": ["G3", "G4"]}, @@ -233,7 +233,8 @@ "depends_on": ["M3"], "status": "in_progress", "acceptance_section": "M5 — Deliver a bounded engineering change", - "open_review_items": ["F2", "G4"], + "open_review_items": ["G4"], + "review_resolution": "F1/R1 and F2/R2 source/offline repairs are verified; R6 check coverage and R8 installed/live proof remain open.", "evidence": [ { "kind": "implementation_checkpoint", @@ -290,7 +291,7 @@ "availability": "tracked_tests" } ], - "blocker": "R1 source repair is verified; independent R2/R6 repairs remain available. The later genuine Claude-to-Codex delivery gate also requires normal Claude login; fixtures cannot substitute for that receipt." + "blocker": "R1/R2 source repairs are verified; independent R6 normal-command check coverage remains. The genuine installed Claude-to-Codex delivery gate also requires normal Claude login; fixtures cannot substitute for that receipt." }, { "id": "M6", diff --git a/docs/plans/engineering-team/evidence/R2-observed-identity-2026-10-01.json b/docs/plans/engineering-team/evidence/R2-observed-identity-2026-10-01.json new file mode 100644 index 0000000..5fc8724 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R2-observed-identity-2026-10-01.json @@ -0,0 +1,88 @@ +{ + "schema_version": 1, + "work_package": "R2", + "status": "source_verified", + "implementation_revision": "ef988897e2dda495482dd26e7c583bc428086ba0", + "recorded_on": "2026-10-01", + "scope": "Claude native reported-model identity, actual-attempt import binding, different-model delivery acceptance and durable failed evidence; offline proof only", + "baseline_reproduction": { + "revision": "6848f11", + "artifact": "R2-identity-red-baseline-2026-09-29.json", + "result": "16 worker tests exposed 39 assertion/subtest failures and four errors before the repair. The original red artifact is preserved." + }, + "acceptance_mapping": [ + { + "requirement": "Concrete reported identity and supported family aliases are validated; missing, wrong, malformed or multiple native model identities fail closed; unreported effort/revision stays unknown", + "tests": "test/core/test_claude_identity.py: 16 tests" + }, + { + "requirement": "Strict JSON, typed native metadata and session/usage/provenance binding reject tampered evidence before candidate finalization", + "tests": "test/core/test_implementation_identity_evidence.py: 12 tests including the exact-attempt and native-byte regressions" + }, + { + "requirement": "The imported selected profile must equal the durable attempt's frozen profile, not merely another allowed fallback", + "tests": "test_implementation_identity_evidence.test_import_must_match_actual_attempt_not_another_allowed_fallback" + }, + { + "requirement": "Invalid UTF-8 and CRLF output retain their original native byte hashes; invalid stderr cannot preempt a valid result", + "tests": "test_implementation_identity_evidence.test_invalid_utf8_stdout_retains_original_native_byte_hash, test_failed_result_hash_preserves_crlf_native_bytes, test_invalid_utf8_stderr_cannot_preempt_valid_result_parsing" + }, + { + "requirement": "Exact and alias-resolved identities complete public delivery; host and headless acceptance reject the same reported model behind different labels", + "tests": "test/core/test_delivery_identity_runtime.py: exact/alias positive cases and host/headless independence cases" + }, + { + "requirement": "Invalid identity cannot publish a candidate; failed diagnostics, native usage and stream references survive fallback, cancellation and replay", + "tests": "test_delivery_identity_runtime.test_invalid_identity_never_publishes_candidate_and_retains_failed_evidence, test_native_fallback_keeps_failed_identity_and_usage_in_success_receipt, test_cancellation_keeps_prior_failed_native_identity_and_usage" + }, + { + "requirement": "Historical v1 receipts remain readable and exact completed acceptance replays; unverified old evidence cannot authorize new acceptance", + "tests": "test_delivery_identity_runtime.test_legacy_identity_cannot_authorize_new_acceptance_but_rejection_replays and test_historical_successful_legacy_receipt_reads_and_exact_acceptance_replays" + } + ], + "verification": { + "core": { + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -v", + "result": "OK (skipped=2)", + "tests_run": 367, + "failures": 0, + "errors": 0, + "skipped": 2, + "suite_seconds": 215.886, + "wrapper_wall_seconds": 216.021, + "wrapper_monotonic_seconds": 216.020, + "exit_code": 0, + "notes": "The command ran inside a subprocess timing wrapper. The source stayed unchanged during discovery. Two official MCP SDK conformance tests skipped because the optional SDK is absent in this interpreter." + }, + "bash": "bash test/run.sh: 11 test files, 227 assertions passed before the planning/evidence checkpoint", + "generated_reference": "python3 scripts/generate-core-reference.py --check: current", + "independent_review": "A bounded source audit found actual-attempt profile binding and exact native-byte retention gaps. Both were reproduced and repaired; a follow-up independently ran four targeted regressions successfully and reported no additional actionable finding in those fixes.", + "earlier_runs": [ + "An earlier 363-test integration run passed with two optional-SDK skips before the last two audit fixes; it is not final-source proof.", + "The first final-source 367-test run completed in 230.999 seconds with six failures and two skips. Four failures showed preflight budget expiry or approximately 15-17 minute UTC gaps, supporting a timing explanation but not establishing the cause of all six failures.", + "All six failing cases reran unchanged successfully in 18.998 seconds with per-test UTC and monotonic elapsed measurements agreeing. The subsequent complete 367-test gate above passed." + ], + "warning": "An ignored finalizer exception reported an unclosed SQLite connection during test_running_cancel_race_reaps_runner_and_worker_once. PYTHONWARNINGS=error::ResourceWarning did not turn that finalizer exception into a failing suite. Cleanup remains R5; this result is not warning-free." + }, + "source_sha256": { + "plugin/core/src/devsquad/claude_identity.py": "765c09b06d9005e88ce14ad735ff033ace88e29bd9c2c5abb77bb45c29edc8c7", + "plugin/core/src/devsquad/claude_delivery_worker.py": "62493f26927c8bd24bce2b39a9ac604290d5fb656f5eeac428eb982a24367450", + "plugin/core/src/devsquad/workflows.py": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "plugin/core/src/devsquad/supervisor.py": "b9732da56ec43cb9235c36ed19c9b982bd81674b4c04dfddba9d76ae9af8d89f", + "plugin/core/src/devsquad/service.py": "07671078c8ac337778ac1e0ba10b1475074b5dfcb3cd0bad030b44d5d54e86c2", + "plugin/core/src/devsquad/reports.py": "df30017e12cb2fd4cd1b39576bb5a2296365ce0611b961ab3a7c437d6d71fef6" + }, + "test_sha256": { + "test/core/test_claude_identity.py": "c18f2f91ff547f2e3c90902bb3516550e20c39e8c9a01f7c8b8ec3ad8f2a2d99", + "test/core/test_implementation_identity_evidence.py": "42ea920c2c31a3df2fd10477d1521f789a69bfbd5ee381fdd20f88e2ad98515c", + "test/core/test_delivery_identity_runtime.py": "b107f3a1c5b4a25b8d9c21bcf76f42d19598ea08b631eebd12fda8b1a6ba044a" + }, + "limitations": [ + "This checkpoint records already implemented source and completed offline verification; the planning update made no production source changes.", + "No real provider generation, installation refresh, authentication reprobe or global provider-setting change occurred. The installed runtime predates R1/R2; affected installed/live proof remains R8.", + "Native modelUsage establishes the bounded reported-model scope only. Unreported effective effort/backing revision remains null; pricing canonicalModel is not serving identity authority; multiple entries do not identify a unique writer.", + "The genuine Claude implementation to different-model review receipt and real Claude host handoff remain blocked on last-observed normal Claude login. No fresh authentication check is claimed.", + "R3-R6 integration, C1/R7, other R8 gates and the SQLite cleanup warning remain open. This is not full M5 or product completion.", + "Raw diagnostics stay outside tracked evidence; this portable record contains only source hashes, test results and bounded observations." + ] +} From 672e38305fcc764d87d7da6afb5914cbf3c5a3ce Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 00:29:02 -0700 Subject: [PATCH 136/197] WIP checkpoint: R3a provenance contracts and fenced prelaunch assignments; integration gate pending (2026-10-01 00:29) --- docs/plans/engineering-team/CONTRACTS.md | 41 +++ docs/plans/engineering-team/M6-STATUS.md | 8 + docs/plans/engineering-team/RESUME.md | 36 ++- .../engineering-team/SOL-REVIEW-FOLLOWUP.md | 11 +- docs/plans/engineering-team/backlog.json | 2 +- .../src/devsquad/experiment_provenance.py | 290 ++++++++++++++++++ plugin/core/src/devsquad/learning.py | 91 +++++- .../migrations/014_experiment_assignments.sql | 29 ++ plugin/core/src/devsquad/store.py | 86 +++++- test/core/test_capacity.py | 2 +- test/core/test_cli.py | 2 +- test/core/test_decision_store.py | 2 +- test/core/test_experiment_assignment_store.py | 197 ++++++++++++ test/core/test_experiment_provenance.py | 261 ++++++++++++++++ test/core/test_experiment_reuse.py | 110 +++++++ test/core/test_experiment_v2_evaluation.py | 212 +++++++++++++ test/core/test_handoff_store.py | 4 +- test/core/test_learning.py | 2 +- test/core/test_lifecycle.py | 2 +- test/core/test_store.py | 8 +- 20 files changed, 1364 insertions(+), 32 deletions(-) create mode 100644 plugin/core/src/devsquad/experiment_provenance.py create mode 100644 plugin/core/src/devsquad/migrations/014_experiment_assignments.sql create mode 100644 test/core/test_experiment_assignment_store.py create mode 100644 test/core/test_experiment_provenance.py create mode 100644 test/core/test_experiment_reuse.py create mode 100644 test/core/test_experiment_v2_evaluation.py diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index fad629f..a7f9467 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -255,3 +255,44 @@ Use this loop: Default experiment budget is disabled until explicitly configured, then at most 10% of eligible runs with a hard call/time cap. Most work uses the current proven policy. V1 uses human-governed static preferences, not exhaustive permutations, an automatic bandit or foundation-model fine-tuning. Report sample sizes and missingness; tiny samples justify hypotheses, not provider rankings. Measure acceptance and critical defects first; also show retries, lead rework, elapsed time, measured usage by pool, blocked time and unmeasured overhead. Final task success and original worker quality are distinct. Pair deterministic checks with review and human correction; a model judging itself is not sufficient evidence. + +### Independent experiment provenance (v2) + +Outcome IDs are globally unique across all cases, arms and splits of a new +evaluation. A profile-binding experiment v2 adds its tested `role` and both +concrete `control_profile_sha256` / `candidate_profile_sha256` values, plus +per-arm `control_execution_sha256` / `candidate_execution_sha256` fingerprints +of the profile and frozen native adapter (including binary/version/provider). +Each +case declares two distinct fingerprints: `case_sha256` identifies the exact +task/candidate corpus item independently of runtime settings, while +`input_sha256` binds the controlled pair context. Corpus identities cannot be +relabelled across evaluation/held-out splits by changing package or policy. + +The corpus projection retains frozen source identity, goal, workflow/task +class, acceptance, checks (including argv/cwd/output declarations), scope and +review mode/focus. Implementer comparisons use the original baseline, never +their produced candidates; reviewer comparisons include the exact candidate. +Pair context additionally retains runtime package, policy, lead mode, budgets, +supporting-role profiles/adapters and all fallback configuration. Only the +explicitly declared tested execution binding and incidental machine paths/origin/runtime capacity +observations are excluded. Absolute path-looking check arguments are not +blanket-removed. Assignment/spec hashes are excluded to avoid a hash cycle. + +An assignment binds the spec hash, project Git common directory, case, split, +arm, tested role/profile/execution fingerprints, both input fingerprints and outcome ID. +Schema 14 records the full spec and original preparation snapshot under the +run's preparation fence, atomically before any attempt. One run cannot be +rebound; one experiment arm cannot acquire a second run. This is an ordering +guarantee, not a comparison of wall-clock timestamps. Plain runs are unchanged. + +Evaluation consumers must derive provenance from those saved records and all +actual relevant attempts. A supplied assignment dictionary or experimental +label is not authority. V2 normalized chains bind final/correction hashes and +distinct run/attempt IDs; an absent partner cannot hide invalid provenance. +Their evidence digest changes when outcomes or correction chains change. +Historical rows remain immutable; current eligibility for a new qualification, +promotion, rollback or fallback must be checked separately from historical +read/replay. Implementation progress and remaining reader/eligibility/public +controller work are tracked in [RESUME.md](RESUME.md), not implied by this +contract or by pure normalized-chain unit fixtures. diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md index 0a0a6a4..08588f4 100644 --- a/docs/plans/engineering-team/M6-STATUS.md +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -8,6 +8,14 @@ catalog and quota connections (G1/G2). Execute R3–R5 in [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The Jev key blocks only its separate measurement; it does not block this engineering work. +R3a now rejects reused outcomes and adds v2 concrete-profile/paired-input +contracts, a stable corpus identity distinct from runtime context, strict +normalized evidence chains and schema-14 prelaunch assignment persistence. +The focused 47-test learning/lifecycle/experiment gate passes. This is partial: +the saved-run v2 reader, shared lifecycle eligibility, historical compatibility +audit and public trial execution are not complete. See RESUME for the current +review/full-gate status; do not treat legacy helper fixtures as public proof. + | Requirement | Planned evidence | Status | |---|---|---| | Strict capacity evidence | Typed pool/window/scope/source/confidence/TTL validation; stale, estimated and incomplete measurements remain unknown | verified at `21b1331` | diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 114af48..64886ca 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,27 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Review correction and next action +R3a is now implemented after `b3de5c6`: global outcome-reuse rejection, +versioned profile/input/assignment contracts, separate corpus-versus-pair +fingerprints, schema-14 prelaunch assignment persistence under the preparation +fence, and strict normalized v2 chain evaluation. The original five reuse +regressions failed before the repair and now pass. The focused experiment, +learning and lifecycle gate currently passes 47 tests; nine targeted migration +checks, including installed-wheel upgrades, also passed. Bounded review found +omitted tested-role fallback policy and native execution identity; both now +have regressions and fixes, including predeclared per-arm execution hashes. +Independent follow-up and the full integration gate remain pending here. + +**R3 remains in progress.** The v2 saved-run outcome reader is not wired into +`Store.evaluate_learning_experiment` yet, so public v2 evaluation fails closed +rather than accepting outcome labels as provenance. R3b must join immutable +assignments, actual attempts and outcomes, then enforce one current-evidence +gate for replay/qualification/promotion/rollback/catalog fallback. Audit/read +compatibility for unsafe legacy v1 proposals and public realistic fixtures +remain R3c; the public paired-trial controller remains R5. Do not count current +legacy positive fixtures as proof of that integration. No live providers, +installation refresh or global settings changes occurred. + R2 source/offline repair is verified at `ef98889`: strict native result parsing, v2 import evidence, actual-model independence and durable failed diagnostics. The 16 original worker regressions, 12 import/parser regressions, 19 existing @@ -32,12 +53,10 @@ recorded; installed/live closure remains R8.** See No live provider call, installed refresh or global provider-setting change occurred. The original red baseline is retained in [R2 baseline evidence](evidence/R2-identity-red-baseline-2026-09-29.json). -The latest user request was to create a plan and feedback for SOL. This -checkpoint updates the canonical handoff and verification records only; no -new implementation was added. SOL should begin **R3a** in -[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), not redo R2 or retry blocked -authentication. R3's two-outcome/three-case promotion bug was reproduced -again without modifying production code; it still needs provenance repair. +The planning handoff was saved at `b3de5c6`; the persistent implementation +goal subsequently resumed R3a as recorded above. Do not redo R2 or retry +blocked authentication. The reproduced two-outcome/three-case reuse is now +rejected, but R3's full saved-run provenance and eligibility repair remains. R1's source repair is verified after `399d93d`: the public regression first reproduced four unsafe mutation paths (review/delivery × host/headless). The @@ -347,8 +366,9 @@ provider paths must not be advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Begin R3a: add the two-outcome/three-case regression and strict frozen - experiment provenance contract. Finish R3b/R3c saved evidence, replay, +2. Complete R3a's pending independent review/integration gate, then implement + the R3b saved-run v2 outcome reader and current-evidence eligibility. Finish + R3b/R3c saved evidence, replay, qualification/promotion/rollback/catalog-fallback eligibility and historical compatibility, then execute R4–R6 in dependency order. Preserve R1/R2, explicit check `output_paths` contract and historical receipts. Do not diff --git a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md index 9c96614..e2bb65b 100644 --- a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md +++ b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md @@ -29,6 +29,13 @@ open under R5; this is not a warning-free or installed/live pass. See [R2 evidence](evidence/R2-observed-identity-2026-10-01.json). Start at **R3**. R3–R8, C1 and affected installed/live proofs remain open. +R3a continuation after `b3de5c6`: outcome uniqueness, the v2 provenance +contract, separate corpus/pair hashes, schema-14 fenced prelaunch assignments +and normalized-chain checks are implemented. Focused tests pass; full gate +and independent patch review are pending. The saved-run v2 outcome reader and +shared lifecycle eligibility are still R3b; do not describe this partial +checkpoint as trustworthy end-to-end qualification or a public trial runner. + ### Ready-to-execute handoff for Sol Do not reimplement R2. It now has a shared strict native parser, v2 imported @@ -458,8 +465,8 @@ permission to claim routing improvement. ## First action for Sol Read the recovery files and verify current Git state. Preserve `ef98889` and -later work. Begin **R3a** with the two-outcome/three-case failing regression and -the versioned provenance contract above, then finish R3b/R3c before R4. +later work. Finish R3a's pending review/full gate, then implement **R3b**'s +saved-run evidence reader and shared current-evidence gate. Complete R3c before R4. Preserve R1's check-integrity gate and R2's actual-attempt/native-identity gate. Continue R3–R6 without waiting on the Claude login or Jev key. Retain R7/R8 and C1 in the full scope. Report each slice as red baseline, verified offline, diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 779cf81..e90ba35 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -24,7 +24,7 @@ "review_work_packages": [ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, - {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "pending", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "checkpoint": "Next implementation: R3a provenance contract and reused-evidence regression, then R3b shared current-evidence eligibility and R3c public/legacy/correction proof. Source mapping is in SOL-REVIEW-FOLLOWUP.md; no R3 implementation is claimed."}, + {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "checkpoint": "R3a outcome uniqueness, v2 profile/execution/input/assignment contracts, schema-14 fenced prelaunch persistence and normalized-chain evaluator implemented after b3de5c6; 47 focused and nine migration tests pass, full gate/review follow-up pending. Reviewed fallback/native-context gaps have regressions and fixes. Public v2 saved-run reader and shared eligibility remain R3b; legacy/public/correction proof remains R3c. R3 is not closed."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, {"id": "R6", "title": "Normal terminal experience and readiness", "status": "pending", "milestones": ["M5", "M7"], "depends_on": ["R1", "R2", "R4", "R5"], "items": ["G3", "G4"]}, diff --git a/plugin/core/src/devsquad/experiment_provenance.py b/plugin/core/src/devsquad/experiment_provenance.py new file mode 100644 index 0000000..167ff40 --- /dev/null +++ b/plugin/core/src/devsquad/experiment_provenance.py @@ -0,0 +1,290 @@ +"""Pure prelaunch experiment contracts; saved-run authority lives in Store. + +An assignment dictionary alone is not execution evidence. It must be frozen +under the preparation fence and checked against actual saved attempts before +it can authorize evaluation or a lifecycle decision. +""" + +from __future__ import annotations + +import hashlib +from pathlib import Path +import re +from typing import Any + +from .contracts import ContractError +from .store import canonical_json +from .validation import validate_profile, validate_task + + +def require_sha256(value: Any, label: str) -> str: + if not isinstance(value, str) or re.fullmatch(r"[0-9a-f]{64}", value) is None: + raise ContractError(f"experiment {label} SHA-256 is invalid") + return value + + +def _digest(value: Any) -> str: + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def assignment_for( + spec: dict[str, Any], case_id: str, arm: str, *, project_common_dir: str, +) -> dict[str, Any]: + """Construct the exact assignment a fenced prelaunch must persist.""" + from .learning import validate_experiment + + spec = validate_experiment(spec) + if spec["schema_version"] != 2: + raise ContractError("experiment assignment requires a v2 specification") + if not isinstance(arm, str) or arm not in {"control", "candidate"}: + raise ContractError("experiment assignment arm is invalid") + if (not isinstance(project_common_dir, str) or not project_common_dir + or not Path(project_common_dir).is_absolute() + or ".." in Path(project_common_dir).parts): + raise ContractError("experiment project common directory is invalid") + case = next((row for row in spec["cases"] if row["case_id"] == case_id), None) + if case is None: + raise ContractError("experiment assignment case is not declared") + variable = spec["variable"] + return { + "schema_version": 1, + "experiment_id": spec["experiment_id"], + "spec_sha256": _digest(spec), + "project_common_dir": project_common_dir, + "case_id": case["case_id"], + "split": case["split"], + "arm": arm, + "role": variable["role"], + "profile_id": variable[f"{arm}_profile_id"], + "profile_sha256": variable[f"{arm}_profile_sha256"], + "execution_sha256": variable[f"{arm}_execution_sha256"], + "input_sha256": case["input_sha256"], + "case_sha256": case["case_sha256"], + "outcome_id": case[f"{arm}_outcome_id"], + } + + +def validate_assignment( + value: dict[str, Any], *, spec: dict[str, Any], project_common_dir: str, +) -> dict[str, Any]: + if not isinstance(value, dict) or type(value.get("schema_version")) is not int: + raise ContractError("experiment assignment fields are invalid") + expected = assignment_for( + spec, value.get("case_id"), value.get("arm"), + project_common_dir=project_common_dir, + ) + if value != expected: + raise ContractError("experiment assignment does not match its frozen contract") + return expected + + +def _profile_identity(selected: Any) -> dict[str, Any]: + if not isinstance(selected, dict) or not isinstance(selected.get("profile"), dict): + raise ContractError("experiment frozen profile is missing") + profile = selected["profile"] + validate_profile(profile) + if (selected.get("profile_id") != profile["id"] + or selected.get("profile_sha256") != _digest(profile)): + raise ContractError("experiment frozen profile fingerprint is inconsistent") + return {"profile_id": profile["id"], "profile_sha256": selected["profile_sha256"]} + + +def _adapter_identity(snapshot: dict[str, Any], role: str, selected: dict[str, Any]) -> Any: + """Preserve supporting native execution context, never auth paths.""" + if selected["profile"]["harness"] == "fixture": + return {"harness": "fixture"} + prefix = {"implementer": "implementation", "reviewer": "review", "lead": "lead"}[role] + adapters = snapshot.get(f"{prefix}_adapters") + if not isinstance(adapters, dict): + raise ContractError("experiment supporting native adapters are missing") + adapter = adapters.get(selected["profile_id"]) + if not isinstance(adapter, dict): + raise ContractError("experiment supporting native adapter is missing") + require_sha256(adapter.get("binary_sha256"), "native binary") + for key in ("harness", "harness_version", "model_provider", "transport"): + if not isinstance(adapter.get(key), str) or not adapter[key]: + raise ContractError("experiment supporting native adapter identity is invalid") + if adapter["harness"] != selected["profile"]["harness"]: + raise ContractError("experiment supporting native adapter harness is inconsistent") + # Frozen adapter schemas already delimit native settings. Exclude only + # resolved machine-local executable and credential locations, not argv, + # tool/permission arguments, protocol/output-schema or binary fingerprints. + return {key: value for key, value in adapter.items() if key not in {"binary", "auth_file"}} + + +def paired_input_identity( + snapshot: dict[str, Any], *, role: str, package_digest: str, +) -> dict[str, str]: + """Hash corpus identity separately from the full controlled pair context. + + No assignment, outcome, produced implementation or mutable runtime field + participates. Thus predeclaring case hashes cannot create a spec/hash cycle. + """ + if not isinstance(snapshot, dict): + raise ContractError("experiment frozen snapshot is missing") + task = snapshot.get("task") + validate_task(task) + if not isinstance(role, str) or role not in {"implementer", "reviewer"}: + raise ContractError("experiment tested role is invalid") + if role == "implementer" and task["workflow"] != "issue-delivery": + raise ContractError("experiment implementer requires issue-delivery") + require_sha256(package_digest, "runtime package") + for key in ("base_oid", "target_oid"): + if (not isinstance(snapshot.get(key), str) + or re.fullmatch(r"(?:[0-9a-f]{40}|[0-9a-f]{64})", snapshot[key]) is None): + raise ContractError("experiment frozen commit identity is invalid") + source = {key: snapshot[key] for key in ("base_oid", "target_oid")} + if role == "reviewer": + workspace = snapshot.get("workspace") + if not isinstance(workspace, dict): + raise ContractError("experiment frozen review candidate is missing") + source["candidate_sha256"] = require_sha256( + workspace.get("candidate_sha256"), "review candidate", + ) + # Delivery review workspaces can have a new candidate target. Record + # their exact commits in addition to the task's original baseline. + for key in ("base_oid", "target_oid"): + value = workspace.get(key) + if not isinstance(value, str) or re.fullmatch(r"(?:[0-9a-f]{40}|[0-9a-f]{64})", value) is None: + raise ContractError("experiment review candidate commit is invalid") + source[f"candidate_{key}"] = value + else: + delivery = snapshot.get("delivery_workspace") + if not isinstance(delivery, dict) or delivery.get("baseline_oid") != snapshot["target_oid"]: + raise ContractError("experiment frozen implementation baseline is inconsistent") + routing = snapshot.get("routing") + if not isinstance(routing, dict) or not isinstance(routing.get("roles"), dict): + raise ContractError("experiment frozen routing is missing") + expected_roles = {"reviewer"} + if task["workflow"] == "issue-delivery": + expected_roles.add("implementer") + if task["lead"]["mode"] == "headless": + expected_roles.add("lead") + if set(routing["roles"]) != expected_roles or role not in expected_roles: + raise ContractError("experiment frozen roles do not match the task") + selected_execution_fingerprint(snapshot, role=role) + try: + policy_sha256 = require_sha256(snapshot["configs"]["policy_file"]["sha256"], "policy") + if policy_sha256 != routing["policy"]["sha256"]: + raise ContractError("experiment frozen policy fingerprint is inconsistent") + except (KeyError, TypeError) as exc: + raise ContractError("experiment frozen policy is missing") from exc + supporting = {} + tested_fallbacks = [] + for name, route in routing["roles"].items(): + if (not isinstance(route, dict) or not isinstance(route.get("selected"), dict) + or not isinstance(route.get("fallbacks"), list) + or not isinstance(route.get("fallback_mode"), str) + or route["fallback_mode"] not in {"none", "policy"} + or (route["fallback_mode"] == "none" and route["fallbacks"])): + raise ContractError("experiment frozen role selection is invalid") + identities = [] + for index, selected in enumerate([route["selected"], *route["fallbacks"]]): + identity = _profile_identity(selected) + if name != role or index > 0: + context_identity = {**identity, "adapter": _adapter_identity(snapshot, name, selected)} + identities.append(context_identity) + if name == role: + tested_fallbacks.append(context_identity) + if name != role: + supporting[name] = {"fallback_mode": route["fallback_mode"], "profiles": identities} + corpus = { + "schema_version": 1, + "source": source, + "task": {key: task[key] for key in ( + "workflow", "goal", "task_class", "acceptance", "checks", "scope", + )}, + "review": task.get("review", {"mode": "standard"}), + } + context = { + "schema_version": 1, "case_sha256": _digest(corpus), + "tested_role": role, "lead": task["lead"], "budget": task["budget"], + "policy_sha256": policy_sha256, "package_digest": package_digest, + "supporting_roles": supporting, + "tested_fallback_mode": routing["roles"][role]["fallback_mode"], + "tested_fallbacks": tested_fallbacks, + } + # Canonical hashing rejects non-finite/non-serializable selected context; + # only these digests cross the evidence boundary. + return {"case_sha256": context["case_sha256"], "input_sha256": _digest(context)} + + +def selected_execution_fingerprint(snapshot: dict[str, Any], *, role: str) -> str: + """Per-arm concrete execution variable, including native version/binary. + + The experiment explicitly declares one fingerprint for each arm. This + permits an intentional harness change without silently permitting drift. + """ + if not isinstance(role, str) or role not in {"implementer", "reviewer"}: + raise ContractError("experiment tested execution role is invalid") + try: + selected = snapshot["routing"]["roles"][role]["selected"] + except (KeyError, TypeError) as exc: + raise ContractError("experiment tested execution selection is missing") from exc + profile = _profile_identity(selected) + return _digest({"profile_sha256": profile["profile_sha256"], + "adapter": _adapter_identity(snapshot, role, selected)}) + + +def validate_arm_chain( + chain: Any, *, spec: dict[str, Any], case: dict[str, Any], arm: str, + project_common_dir: str, evaluated_at: str, +) -> dict[str, Any]: + """Check a normalized chain supplied by the saved-run evidence reader. + + These shape/hash checks do not authorize importing arbitrary caller + provenance. Store must derive the witness from fenced prelaunch records, + actual attempts and append-only outcomes, not accept a submitted witness. + """ + from .learning import _timestamp, validate_outcome + + if not isinstance(chain, dict) or set(chain) != {"final", "late_corrections", "provenance"}: + raise ContractError("experiment arm requires saved-run provenance") + provenance = chain["provenance"] + fields = { + "run_id", "assignment", "profile_sha256", "input_sha256", "case_sha256", + "attempt_ids", "attempts_sha256", "final_outcome_sha256", "correction_sha256", + "execution_sha256", + } + if not isinstance(provenance, dict) or set(provenance) != fields: + raise ContractError("experiment arm provenance fields are invalid") + expected = assignment_for(spec, case["case_id"], arm, project_common_dir=project_common_dir) + if validate_assignment(provenance["assignment"], spec=spec, project_common_dir=project_common_dir) != expected: + raise ContractError("experiment arm provenance has a crossed assignment") + if any(provenance[key] != expected[key] for key in ( + "profile_sha256", "execution_sha256", "input_sha256", "case_sha256", + )): + raise ContractError("experiment arm provenance does not match actual profile/input") + run_id = provenance["run_id"] + if not isinstance(run_id, str) or not run_id.strip(): + raise ContractError("experiment provenance run id is missing") + attempt_ids = provenance["attempt_ids"] + if (not isinstance(attempt_ids, list) or not attempt_ids + or any(not isinstance(item, str) or not item.strip() for item in attempt_ids) + or len(set(attempt_ids)) != len(attempt_ids)): + raise ContractError("experiment provenance requires distinct actual attempt ids") + require_sha256(provenance["attempts_sha256"], "actual attempts") + final = validate_outcome(chain["final"], now=_timestamp(evaluated_at, "evaluated_at")) + if (final["kind"] != "final" or final["selection_mode"] != "experimental" + or final["outcome_id"] != expected["outcome_id"]): + raise ContractError("experiment provenance final outcome is inconsistent") + if provenance["final_outcome_sha256"] != _digest(final): + raise ContractError("experiment provenance final outcome hash is inconsistent") + corrections = chain["late_corrections"] + if not isinstance(corrections, list): + raise ContractError("experiment provenance corrections must be an array") + correction_ids = set() + hashes = [] + for correction in corrections: + normalized = validate_outcome(correction, now=_timestamp(evaluated_at, "evaluated_at")) + if (normalized["kind"] != "late_correction" or normalized["selection_mode"] != "experimental" + or normalized["corrects_outcome_id"] != final["outcome_id"] + or _timestamp(normalized["observed_at"], "observed_at") + < _timestamp(final["observed_at"], "observed_at") + or normalized["outcome_id"] in correction_ids): + raise ContractError("experiment provenance correction chain is inconsistent") + correction_ids.add(normalized["outcome_id"]) + hashes.append(_digest(normalized)) + if provenance["correction_sha256"] != hashes: + raise ContractError("experiment provenance correction hashes are inconsistent") + return provenance diff --git a/plugin/core/src/devsquad/learning.py b/plugin/core/src/devsquad/learning.py index 0f1ca12..4d34f0d 100644 --- a/plugin/core/src/devsquad/learning.py +++ b/plugin/core/src/devsquad/learning.py @@ -344,7 +344,7 @@ def validate_experiment(value: dict[str, Any]) -> dict[str, Any]: """Validate a predeclared one-variable, paired outcome experiment.""" if not isinstance(value, dict) or set(value) != EXPERIMENT_FIELDS: raise ContractError("experiment fields are invalid") - if type(value["schema_version"]) is not int or value["schema_version"] != 1: + if type(value["schema_version"]) is not int or value["schema_version"] not in {1, 2}: raise ContractError("experiment schema_version is invalid") experiment_id = _identifier(value["experiment_id"], "experiment_id") project_path = value["project_path"] @@ -360,9 +360,15 @@ def validate_experiment(value: dict[str, Any]) -> dict[str, Any]: }): raise ContractError("experiment evidence_availability is invalid") variable = value["variable"] - if not isinstance(variable, dict) or set(variable) != { + variable_fields = { "kind", "alias", "control_profile_id", "candidate_profile_id", - }: + } + if value["schema_version"] == 2: + variable_fields.update({ + "role", "control_profile_sha256", "candidate_profile_sha256", + "control_execution_sha256", "candidate_execution_sha256", + }) + if not isinstance(variable, dict) or set(variable) != variable_fields: raise ContractError("experiment variable fields are invalid") if variable["kind"] != "profile_binding": raise ContractError("experiment variable kind is invalid") @@ -370,6 +376,15 @@ def validate_experiment(value: dict[str, Any]) -> dict[str, Any]: _identifier(variable[field], f"variable.{field}") if variable["control_profile_id"] == variable["candidate_profile_id"]: raise ContractError("experiment control and candidate must differ") + if value["schema_version"] == 2: + from .experiment_provenance import require_sha256 + + if (not isinstance(variable["role"], str) + or variable["role"] not in {"implementer", "reviewer"}): + raise ContractError("experiment tested role is invalid") + for arm in ("control", "candidate"): + require_sha256(variable[f"{arm}_profile_sha256"], f"{arm} profile") + require_sha256(variable[f"{arm}_execution_sha256"], f"{arm} execution") budget = value["budget"] if not isinstance(budget, dict) or set(budget) != { @@ -386,12 +401,17 @@ def validate_experiment(value: dict[str, Any]) -> dict[str, Any]: if not isinstance(cases, list) or not cases or len(cases) > budget["max_cases"]: raise ContractError("experiment cases exceed the bounded case budget") case_ids = set() + outcome_ids = set() + case_hashes = set() normalized_cases = [] split_counts = {"evaluation": 0, "held_out": 0} for case in cases: - if not isinstance(case, dict) or set(case) != { + case_fields = { "case_id", "split", "control_outcome_id", "candidate_outcome_id", - }: + } + if value["schema_version"] == 2: + case_fields.update({"input_sha256", "case_sha256"}) + if not isinstance(case, dict) or set(case) != case_fields: raise ContractError("experiment case fields are invalid") case_id = _identifier(case["case_id"], "case_id") if case_id in case_ids: @@ -406,7 +426,17 @@ def validate_experiment(value: dict[str, Any]) -> dict[str, Any]: ) if control_id == candidate_id: raise ContractError("experiment paired outcomes must differ") + if control_id in outcome_ids or candidate_id in outcome_ids: + raise ContractError("experiment outcome ids must be globally unique") + outcome_ids.update((control_id, candidate_id)) + if value["schema_version"] == 2: + require_sha256(case["input_sha256"], "paired input") + require_sha256(case["case_sha256"], "corpus case") + if case["case_sha256"] in case_hashes: + raise ContractError("experiment corpus case identities must be unique") + case_hashes.add(case["case_sha256"]) normalized_cases.append({ + **case, "case_id": case_id, "split": case["split"], "control_outcome_id": control_id, @@ -460,12 +490,41 @@ def evaluate_experiment( outcome_chains: dict[str, dict[str, Any]], *, evaluated_at: str, + project_common_dir: str | None = None, ) -> dict[str, Any]: """Evaluate a frozen paired experiment without changing active policy.""" spec = validate_experiment(experiment) _timestamp(evaluated_at, "evaluated_at") if not isinstance(outcome_chains, dict): raise ContractError("experiment outcome chains are invalid") + provenance = {} + if spec["schema_version"] == 2: + from .experiment_provenance import validate_arm_chain + + if not isinstance(project_common_dir, str) or not Path(project_common_dir).is_absolute(): + raise ContractError("experiment provenance requires the saved project identity") + run_ids = set() + attempt_ids = set() + # Validate every available arm, even when its partner is missing. A + # missing partner must not hide reused or crossed evidence. + for case in spec["cases"]: + for arm in ("control", "candidate"): + outcome_id = case[f"{arm}_outcome_id"] + chain = outcome_chains.get(outcome_id) + if chain is None: + provenance[outcome_id] = None + continue + witness = validate_arm_chain( + chain, spec=spec, case=case, arm=arm, + project_common_dir=project_common_dir, evaluated_at=evaluated_at, + ) + if witness["run_id"] in run_ids: + raise ContractError("experiment provenance run ids must be globally unique") + if attempt_ids.intersection(witness["attempt_ids"]): + raise ContractError("experiment provenance attempt ids must be globally unique") + run_ids.add(witness["run_id"]) + attempt_ids.update(witness["attempt_ids"]) + provenance[outcome_id] = witness rows = [] metrics = { split: { @@ -497,7 +556,10 @@ def evaluate_experiment( rows.append(row) continue for arm, chain in (("control", control), ("candidate", candidate)): - if (not isinstance(chain, dict) or set(chain) != {"final", "late_corrections"} + chain_fields = {"final", "late_corrections"} + if spec["schema_version"] == 2: + chain_fields.add("provenance") + if (not isinstance(chain, dict) or set(chain) != chain_fields or not isinstance(chain["final"], dict) or not isinstance(chain["late_corrections"], list)): raise ContractError("experiment outcome chain is invalid") @@ -555,8 +617,8 @@ def evaluate_experiment( reasons.append("candidate_escaped_defect_limit_exceeded") verdict = "promotion_proposal" if not reasons else "no_change" - return { - "schema_version": 1, + evaluation = { + "schema_version": spec["schema_version"], "experiment_id": spec["experiment_id"], "spec_sha256": hashlib.sha256( canonical_json(spec).encode(), @@ -572,6 +634,9 @@ def evaluate_experiment( "rollback_target": spec["rollback_target"], "evidence_availability": spec["evidence_availability"], } + if spec["schema_version"] == 2: + evaluation["evidence_sha256"] = hashlib.sha256(canonical_json(provenance).encode()).hexdigest() + return evaluation def build_learning_proposal( @@ -624,9 +689,17 @@ def build_learning_proposal( raise ContractError("learning proposal experiment record is invalid") spec = validate_experiment(experiment_record["experiment"]) evaluation = experiment_record["evaluation"] + expected_fields = EXPERIMENT_EVALUATION_FIELDS | ( + {"evidence_sha256"} if spec["schema_version"] == 2 else set() + ) if (not isinstance(evaluation, dict) - or set(evaluation) != EXPERIMENT_EVALUATION_FIELDS): + or set(evaluation) != expected_fields + or type(evaluation.get("schema_version")) is not int + or evaluation["schema_version"] != spec["schema_version"]): raise ContractError("learning proposal evaluation is invalid") + if spec["schema_version"] == 2: + from .experiment_provenance import require_sha256 + require_sha256(evaluation["evidence_sha256"], "evaluated evidence") spec_sha256 = hashlib.sha256(canonical_json(spec).encode()).hexdigest() evaluation_sha256 = hashlib.sha256( canonical_json(evaluation).encode(), diff --git a/plugin/core/src/devsquad/migrations/014_experiment_assignments.sql b/plugin/core/src/devsquad/migrations/014_experiment_assignments.sql new file mode 100644 index 0000000..2dbf0f8 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/014_experiment_assignments.sql @@ -0,0 +1,29 @@ +-- Predeclared experiments and arms are frozen by the run preparation fence, +-- before attempt reservation. Existing evaluations remain immutable history. +CREATE TABLE experiment_specs ( + experiment_id TEXT PRIMARY KEY, + project_id TEXT NOT NULL REFERENCES projects(id), + spec_json TEXT NOT NULL, + spec_sha256 TEXT NOT NULL + CHECK(length(spec_sha256) = 64 AND spec_sha256 NOT GLOB '*[^0-9a-f]*'), + recorded_at TEXT NOT NULL +); + +CREATE TABLE experiment_assignments ( + run_id TEXT PRIMARY KEY REFERENCES runs(id), + experiment_id TEXT NOT NULL REFERENCES experiment_specs(experiment_id), + case_id TEXT NOT NULL, + arm TEXT NOT NULL CHECK(arm IN ('control', 'candidate')), + outcome_id TEXT NOT NULL, + assignment_json TEXT NOT NULL, + assignment_sha256 TEXT NOT NULL + CHECK(length(assignment_sha256) = 64 AND assignment_sha256 NOT GLOB '*[^0-9a-f]*'), + snapshot_json TEXT NOT NULL, + package_digest TEXT NOT NULL + CHECK(length(package_digest) = 64 AND package_digest NOT GLOB '*[^0-9a-f]*'), + frozen_run_version INTEGER NOT NULL, + preparation_fencing_token INTEGER NOT NULL, + recorded_at TEXT NOT NULL, + UNIQUE(experiment_id, case_id, arm), + UNIQUE(experiment_id, outcome_id) +); diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index d049d06..bc1baf3 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 13 +SUPPORTED_SCHEMA_VERSION = 14 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -409,6 +409,9 @@ def complete_preparation( if not predecessor or predecessor["project_id"] != project["project_id"] or predecessor["state"] not in TERMINAL_STATES: raise ConflictError("superseded run must be terminal and belong to the same project") version, now = row["version"] + 1, _utc_now() + self._freeze_experiment_assignment( + run_id, fencing_token, version, mutable_snapshot, package_digest, now, + ) self.connection.execute( "UPDATE runs SET mutable_snapshot=?,package_path=?,package_digest=?," "supersedes_run_id=?,worktree_path=COALESCE(?,worktree_path),phase=NULL," @@ -424,6 +427,87 @@ def complete_preparation( self.connection.execute("ROLLBACK") raise + def _freeze_experiment_assignment( + self, run_id: str, fencing_token: int, run_version: int, + snapshot: Any, package_digest: str | None, recorded_at: str, + ) -> None: + """Inside complete_preparation's transaction, before any launch is possible.""" + from .experiment_provenance import ( + paired_input_identity, selected_execution_fingerprint, validate_assignment, + ) + from .learning import validate_experiment + + if not isinstance(snapshot, dict): + return + fields = {"experiment_spec", "experiment_assignment"} & set(snapshot) + if not fields: + return + if fields != {"experiment_spec", "experiment_assignment"}: + raise ContractError("experiment preparation requires both specification and assignment") + spec = validate_experiment(snapshot["experiment_spec"]) + project = self.connection.execute( + "SELECT r.project_id,p.git_common_dir FROM runs r " + "JOIN projects p ON p.id=r.project_id WHERE r.id=?", (run_id,), + ).fetchone() + if str(git_common_dir(Path(spec["project_path"]))) != project["git_common_dir"]: + raise ContractError("experiment specification belongs to another project") + assignment = validate_assignment( + snapshot["experiment_assignment"], spec=spec, + project_common_dir=project["git_common_dir"], + ) + identity = paired_input_identity( + snapshot, role=assignment["role"], package_digest=package_digest, + ) + if any(identity[key] != assignment[key] for key in identity): + raise ContractError("experiment paired input does not match its declaration") + selected = snapshot["routing"]["roles"][assignment["role"]]["selected"] + if any(selected[key] != assignment[key] for key in ("profile_id", "profile_sha256")): + raise ContractError("experiment selected profile does not match its arm") + if selected_execution_fingerprint(snapshot, role=assignment["role"]) != assignment["execution_sha256"]: + raise ContractError("experiment native execution does not match its declared arm") + if self.connection.execute( + "SELECT 1 FROM attempts WHERE run_id=? LIMIT 1", (run_id,), + ).fetchone() is not None: + raise ConflictError("experiment assignment must precede every attempt") + spec_json = canonical_json(spec) + spec_sha256 = hashlib.sha256(spec_json.encode()).hexdigest() + existing = self.connection.execute( + "SELECT project_id,spec_json,spec_sha256 FROM experiment_specs WHERE experiment_id=?", + (spec["experiment_id"],), + ).fetchone() + if existing is not None: + if (existing["project_id"] != project["project_id"] + or existing["spec_json"] != spec_json or existing["spec_sha256"] != spec_sha256): + raise ConflictError("experiment specification is already frozen differently") + else: + # An evaluation created before this contract cannot be retroactively + # upgraded into a predeclared trial with the same identifier. + if self.connection.execute( + "SELECT 1 FROM experiments WHERE experiment_id=?", (spec["experiment_id"],), + ).fetchone() is not None: + raise ConflictError("historical experiment cannot acquire new assignments") + self.connection.execute( + "INSERT INTO experiment_specs(experiment_id,project_id,spec_json,spec_sha256,recorded_at) " + "VALUES(?,?,?,?,?)", + (spec["experiment_id"], project["project_id"], spec_json, spec_sha256, recorded_at), + ) + if self.connection.execute( + "SELECT 1 FROM experiment_assignments WHERE run_id=? OR " + "(experiment_id=? AND (outcome_id=? OR (case_id=? AND arm=?)))", + (run_id, spec["experiment_id"], assignment["outcome_id"], + assignment["case_id"], assignment["arm"]), + ).fetchone() is not None: + raise ConflictError("experiment run or arm is already assigned") + payload = canonical_json(assignment) + self.connection.execute( + "INSERT INTO experiment_assignments(run_id,experiment_id,case_id,arm,outcome_id," + "assignment_json,assignment_sha256,snapshot_json,package_digest,frozen_run_version," + "preparation_fencing_token,recorded_at) VALUES(?,?,?,?,?,?,?,?,?,?,?,?)", + (run_id, spec["experiment_id"], assignment["case_id"], assignment["arm"], + assignment["outcome_id"], payload, hashlib.sha256(payload.encode()).hexdigest(), + canonical_json(snapshot), package_digest, run_version, fencing_token, recorded_at), + ) + def fail_preparation( self, run_id: str, diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index fd6bd9a..a4f421e 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -184,7 +184,7 @@ def test_migration_nine_creates_capacity_ledger(self): version = store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 13) + self.assertEqual(version, 14) tables = { row[0] for row in store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_cli.py b/test/core/test_cli.py index e903188..538cf1b 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -693,7 +693,7 @@ def test_installed_wheel_contains_and_applies_current_migrations(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 13 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 14 attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} diff --git a/test/core/test_decision_store.py b/test/core/test_decision_store.py index 294250e..db9707b 100644 --- a/test/core/test_decision_store.py +++ b/test/core/test_decision_store.py @@ -191,7 +191,7 @@ def test_schema_thirteen_contains_decision_cache_and_run_links(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 13) + self.assertEqual(version, 14) tables = { row[0] for row in self.store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_experiment_assignment_store.py b/test/core/test_experiment_assignment_store.py new file mode 100644 index 0000000..aba47ea --- /dev/null +++ b/test/core/test_experiment_assignment_store.py @@ -0,0 +1,197 @@ +"""Prelaunch assignment persistence, not a claim of public trial execution.""" + +import copy +import hashlib +import json +from pathlib import Path +import sqlite3 +import subprocess +import tempfile +import unittest + +from test_experiment_provenance import frozen_snapshot, spec_v2, digest, execution_digest +from test_learning import experiment +from test_lifecycle import profile +from devsquad.contracts import ContractError +from devsquad.experiment_provenance import assignment_for, paired_input_identity +from devsquad.store import ConflictError, Store, canonical_json, git_common_dir + + +class ExperimentAssignmentStoreTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix='devsquad-assignment-') + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name).resolve() + self.repo = self.root / 'repo' + subprocess.run(['git', 'init', '-q', str(self.repo)], check=True) + self.store = Store(self.root / 'state.sqlite3', self.root / 'artifacts') + self.addCleanup(self.store.close) + self.common = str(git_common_dir(self.repo)) + self.snapshot = frozen_snapshot() + self.snapshot['task']['project']['repo_path'] = str(self.repo) + self.spec = spec_v2() + self.spec['project_path'] = str(self.repo) + self.spec['cases'][0].update(paired_input_identity( + self.snapshot, role='reviewer', package_digest='e' * 64, + )) + + def snapshot_for(self, arm='control', *, spec=None): + spec = spec or self.spec + snapshot = copy.deepcopy(self.snapshot) + if arm == 'candidate': + candidate = profile('profile-b', 'model-b') + snapshot['routing']['roles']['reviewer']['selected'] = { + 'profile_id': candidate['id'], 'profile': candidate, + 'profile_sha256': digest(candidate), + } + snapshot['experiment_spec'] = copy.deepcopy(spec) + snapshot['experiment_assignment'] = assignment_for( + spec, 'eval-1', arm, project_common_dir=self.common, + ) + return snapshot + + def claim(self, key): + return self.store.claim_start(self.repo, key, {'task': self.snapshot['task']}, 'owner') + + def prepare(self, claim, snapshot): + return self.store.complete_preparation( + claim.run_id, claim.fencing_token, snapshot, package_digest='e' * 64, + ) + + def test_assignment_freezes_before_attempts_with_preparation_fence(self): + claim = self.claim('first') + snapshot = self.snapshot_for() + version = self.prepare(claim, snapshot) + row = self.store.connection.execute( + 'SELECT * FROM experiment_assignments WHERE run_id=?', (claim.run_id,), + ).fetchone() + self.assertEqual(json.loads(row['assignment_json']), snapshot['experiment_assignment']) + self.assertEqual(json.loads(row['snapshot_json']), snapshot) + self.assertEqual(row['assignment_sha256'], digest(snapshot['experiment_assignment'])) + self.assertEqual(row['preparation_fencing_token'], claim.fencing_token) + self.assertEqual(row['frozen_run_version'], version) + self.assertEqual(self.store.connection.execute( + 'SELECT COUNT(*) FROM attempts WHERE run_id=?', (claim.run_id,), + ).fetchone()[0], 0) + # Later mutable workflow snapshots cannot rewrite the original witness. + self.store.connection.execute('UPDATE runs SET mutable_snapshot=? WHERE id=?', + (canonical_json({'altered': True}), claim.run_id)) + self.assertEqual(self.store.connection.execute( + 'SELECT snapshot_json FROM experiment_assignments WHERE run_id=?', (claim.run_id,), + ).fetchone()[0], canonical_json(snapshot)) + + def test_distinct_arms_share_one_predeclared_spec(self): + for arm in ('control', 'candidate'): + self.prepare(self.claim(arm), self.snapshot_for(arm)) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_specs').fetchone()[0], 1) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 2) + + def test_invalid_assignment_rolls_back_spec_and_preparation_atomically(self): + for mutation in ('input', 'profile', 'project', 'missing_spec', 'digest'): + with self.subTest(mutation=mutation): + claim = self.claim(mutation) + snapshot = self.snapshot_for() + if mutation == 'input': + snapshot['task']['goal'] = 'A different goal.' + elif mutation == 'profile': + snapshot['experiment_assignment'] = assignment_for( + self.spec, 'eval-1', 'candidate', project_common_dir=self.common, + ) + elif mutation == 'project': + snapshot['experiment_assignment']['project_common_dir'] = '/tmp/foreign/.git' + elif mutation == 'missing_spec': + del snapshot['experiment_spec'] + else: + snapshot['experiment_assignment']['spec_sha256'] = '0' * 64 + with self.assertRaises(ContractError): + self.prepare(claim, snapshot) + self.assertEqual(self.store.run(claim.run_id)['phase'], 'preparing') + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_specs').fetchone()[0], 0) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 0) + + def test_duplicate_arm_cannot_acquire_a_second_run(self): + first = self.claim('first') + self.prepare(first, self.snapshot_for()) + second = self.claim('second') + with self.assertRaisesRegex(ConflictError, 'already assigned'): + self.prepare(second, self.snapshot_for()) + self.assertEqual(self.store.run(second.run_id)['phase'], 'preparing') + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 1) + + def test_changed_spec_and_stale_preparation_cannot_rebind(self): + first = self.claim('first') + self.prepare(first, self.snapshot_for()) + with self.assertRaisesRegex(ConflictError, 'stale'): + self.prepare(first, self.snapshot_for('candidate')) + changed = copy.deepcopy(self.spec) + changed['hypothesis'] = 'Post-hoc replacement.' + second = self.claim('candidate') + with self.assertRaisesRegex(ConflictError, 'frozen differently'): + self.prepare(second, self.snapshot_for('candidate', spec=changed)) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 1) + + def test_plain_preparation_creates_no_experiment_records(self): + claim = self.claim('plain') + self.prepare(claim, self.snapshot) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_specs').fetchone()[0], 0) + with self.assertRaisesRegex(ConflictError, 'stale'): + self.prepare(claim, self.snapshot_for()) + + def test_tested_native_version_and_binary_must_match_predeclared_execution(self): + snapshot = copy.deepcopy(self.snapshot) + selected = snapshot['routing']['roles']['reviewer']['selected'] + selected['profile']['harness'] = 'codex' + selected['profile_sha256'] = digest(selected['profile']) + adapter = {'harness': 'codex', 'harness_version': 'fixture-version-1', + 'transport': 'native_protocol', 'model_provider': 'openai', + 'binary_sha256': '1' * 64} + snapshot['review_adapters'] = {'profile-a': adapter} + spec = copy.deepcopy(self.spec) + spec['variable']['control_profile_sha256'] = selected['profile_sha256'] + spec['variable']['control_execution_sha256'] = execution_digest(selected['profile'], adapter) + snapshot['experiment_spec'] = spec + snapshot['experiment_assignment'] = assignment_for( + spec, 'eval-1', 'control', project_common_dir=self.common, + ) + for field, replacement in (('harness_version', 'fixture-version-2'), + ('binary_sha256', '2' * 64)): + changed = copy.deepcopy(snapshot) + changed['review_adapters']['profile-a'][field] = replacement + with self.subTest(field=field): + with self.assertRaisesRegex(ContractError, 'native execution'): + self.prepare(self.claim(field), changed) + # Correct explicit native context is accepted; no CLI is executed. + self.prepare(self.claim('native-context'), snapshot) + self.assertEqual(self.store.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 1) + + def test_schema_13_migration_preserves_legacy_evaluation_bytes(self): + database = self.root / 'old.sqlite3' + connection = sqlite3.connect(database) + migrations = Path(__file__).resolve().parents[2] / 'plugin/core/src/devsquad/migrations' + for path in sorted(migrations.glob('*.sql')): + version = int(path.name.split('_', 1)[0]) + if version > 13: + continue + connection.executescript(path.read_text()) + connection.execute('INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)', (version, 'fixture')) + spec = canonical_json(experiment(self.repo)) + evaluation = canonical_json({'schema_version': 1, 'historical': True}) + connection.execute( + 'INSERT INTO experiments(experiment_id,project_path,spec_json,spec_sha256,evaluation_json,' + 'evaluation_sha256,verdict,recorded_at) VALUES(?,?,?,?,?,?,?,?)', + ('old', str(self.repo), spec, hashlib.sha256(spec.encode()).hexdigest(), + evaluation, hashlib.sha256(evaluation.encode()).hexdigest(), 'no_change', 'fixture'), + ) + connection.commit() + connection.close() + upgraded = Store(database, self.root / 'old-artifacts') + self.addCleanup(upgraded.close) + saved = upgraded.connection.execute('SELECT * FROM experiments').fetchone() + self.assertEqual(saved['spec_json'], spec) + self.assertEqual(saved['evaluation_json'], evaluation) + self.assertEqual(upgraded.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0], 14) + self.assertEqual(upgraded.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 0) + + +if __name__ == '__main__': + unittest.main() diff --git a/test/core/test_experiment_provenance.py b/test/core/test_experiment_provenance.py new file mode 100644 index 0000000..b9777c6 --- /dev/null +++ b/test/core/test_experiment_provenance.py @@ -0,0 +1,261 @@ +"""Versioned experiment assignment and paired-input identity contracts.""" + +import copy +import hashlib +import importlib +from pathlib import Path +import unittest + +from test_learning import NOW, experiment, experimental_final +from test_lifecycle import profile, review_task +from devsquad.contracts import ContractError +from devsquad.learning import evaluate_experiment, validate_experiment +from devsquad.store import canonical_json + + +def digest(value): + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def execution_digest(selected, adapter=None): + return digest({'profile_sha256': digest(selected), + 'adapter': adapter if adapter is not None else {'harness': 'fixture'}}) + + +def spec_v2(): + spec = experiment(Path('/tmp/provenance-project')) + spec['schema_version'] = 2 + spec['variable'].update({ + 'role': 'reviewer', + 'control_profile_sha256': digest(profile('profile-a', 'model-a')), + 'candidate_profile_sha256': digest(profile('profile-b', 'model-b')), + 'control_execution_sha256': execution_digest(profile('profile-a', 'model-a')), + 'candidate_execution_sha256': execution_digest(profile('profile-b', 'model-b')), + }) + for case in spec['cases']: + case['input_sha256'] = digest(['input', case['case_id']]) + case['case_sha256'] = digest(['case', case['case_id']]) + return spec + + +def frozen_snapshot(): + selected = profile('profile-a', 'model-a') + return { + 'task': review_task(Path('/tmp/provenance-project')), + 'base_oid': 'a' * 40, + 'target_oid': 'b' * 40, + 'workspace': { + 'candidate_sha256': 'c' * 64, 'path': '/tmp/run-a/review', + 'base_oid': 'a' * 40, 'target_oid': 'b' * 40, + }, + 'configs': {'policy_file': {'path': '/tmp/a/policy.json', 'sha256': 'd' * 64}}, + 'routing': { + 'policy': {'sha256': 'd' * 64}, + 'roles': {'reviewer': { + 'selected': {'profile_id': selected['id'], 'profile': selected, + 'profile_sha256': digest(selected)}, + 'fallbacks': [], 'fallback_mode': 'none', + }}, + 'capacity': {}, + }, + } + + +class ExperimentProvenanceContractTest(unittest.TestCase): + def api(self): + return importlib.import_module('devsquad.experiment_provenance') + + def test_v2_spec_retains_concrete_profiles_and_both_input_identities(self): + spec = spec_v2() + self.assertEqual(validate_experiment(spec), spec) + self.assertEqual(spec['variable']['role'], 'reviewer') + + def test_v2_spec_requires_role_fingerprints_and_exact_case_fields(self): + changes = [ + ('variable', 'role', 'lead'), + ('variable', 'role', []), + ('variable', 'control_profile_sha256', 'short'), + ('variable', 'candidate_profile_sha256', True), + ('variable', 'control_execution_sha256', None), + ('variable', 'candidate_execution_sha256', 'short'), + ('case', 'case_sha256', None), + ('case', 'input_sha256', 'A' * 64), + ] + for target, field, value in changes: + with self.subTest(field=field, value=value): + spec = spec_v2() + row = spec['variable'] if target == 'variable' else spec['cases'][0] + row[field] = value + with self.assertRaises(ContractError): + validate_experiment(spec) + for field in ('role', 'control_profile_sha256', 'candidate_profile_sha256', + 'control_execution_sha256', 'candidate_execution_sha256'): + spec = spec_v2() + del spec['variable'][field] + with self.assertRaises(ContractError): + validate_experiment(spec) + + def test_v2_cannot_evaluate_unbound_outcome_labels_as_provenance(self): + spec = spec_v2() + chains = {} + for case in spec['cases']: + for arm, verdict in (('control', 'failed'), ('candidate', 'succeeded')): + outcome_id = case[f'{arm}_outcome_id'] + chains[outcome_id] = { + 'final': experimental_final(outcome_id, verdict), 'late_corrections': [], + } + with self.assertRaisesRegex(ContractError, 'provenance'): + evaluate_experiment(spec, chains, evaluated_at=NOW.isoformat()) + + def test_same_corpus_case_cannot_be_relabelled_held_out_by_context_change(self): + spec = spec_v2() + spec['cases'][2]['case_sha256'] = spec['cases'][0]['case_sha256'] + self.assertNotEqual(spec['cases'][2]['input_sha256'], spec['cases'][0]['input_sha256']) + with self.assertRaisesRegex(ContractError, 'case.*unique'): + validate_experiment(spec) + + def test_assignment_is_exactly_bound_to_spec_case_split_arm_and_project(self): + api = self.api() + spec = spec_v2() + assignment = api.assignment_for( + spec, 'eval-1', 'candidate', project_common_dir='/tmp/provenance-project/.git', + ) + self.assertEqual(assignment['spec_sha256'], digest(spec)) + self.assertEqual(assignment['profile_sha256'], spec['variable']['candidate_profile_sha256']) + self.assertEqual(api.validate_assignment( + assignment, spec=spec, project_common_dir='/tmp/provenance-project/.git', + ), assignment) + for field, replacement in ( + ('schema_version', True), ('spec_sha256', '0' * 64), + ('experiment_id', 'foreign-experiment'), ('split', 'held_out'), + ('arm', 'control'), ('role', 'implementer'), + ('profile_id', 'profile-a'), ('profile_sha256', '0' * 64), + ('execution_sha256', '0' * 64), + ('input_sha256', '0' * 64), ('case_sha256', '0' * 64), + ('project_common_dir', '/tmp/foreign/.git'), ('extra', 1), + ): + with self.subTest(field=field): + invalid = {**assignment, field: replacement} + with self.assertRaises(ContractError): + api.validate_assignment(invalid, spec=spec, + project_common_dir='/tmp/provenance-project/.git') + + def test_legacy_specs_and_unknown_case_or_arm_cannot_make_assignments(self): + api = self.api() + for spec, case_id, arm in ( + (experiment(Path('/tmp/provenance-project')), 'eval-1', 'candidate'), + (spec_v2(), 'missing', 'candidate'), + (spec_v2(), 'eval-1', []), + ): + with self.subTest(case=case_id, arm=arm): + with self.assertRaises(ContractError): + api.assignment_for(spec, case_id, arm, + project_common_dir='/tmp/provenance-project/.git') + + def test_incidental_paths_observations_and_tested_profile_do_not_change_inputs(self): + api = self.api() + before = frozen_snapshot() + changed = copy.deepcopy(before) + changed['task']['project']['repo_path'] = '/tmp/other-worktree' + changed['task']['project']['base_ref'] = 'renamed-base' + changed['task']['origin'] = {'surface': 'another-host', 'session_ref': 'session-b'} + changed['workspace']['path'] = '/tmp/run-b/review' + changed['configs']['policy_file']['path'] = '/tmp/b/policy.json' + changed['routing']['capacity'] = {'irrelevant': 'runtime observation'} + candidate = profile('profile-b', 'model-b') + changed['routing']['roles']['reviewer']['selected'] = { + 'profile_id': candidate['id'], 'profile': candidate, + 'profile_sha256': digest(candidate), + } + # Assignment is deliberately excluded: its spec hash includes input hashes. + changed['experiment_assignment'] = {'not': 'fingerprint input'} + self.assertEqual(api.paired_input_identity(before, role='reviewer', package_digest='e' * 64), + api.paired_input_identity(changed, role='reviewer', package_digest='e' * 64)) + + def test_runtime_policy_and_peer_changes_affect_pair_not_corpus_identity(self): + api = self.api() + before = frozen_snapshot() + base = api.paired_input_identity(before, role='reviewer', package_digest='e' * 64) + for kind in ('runtime', 'policy', 'peer'): + with self.subTest(kind=kind): + changed = copy.deepcopy(before) + package = 'e' * 64 + if kind == 'runtime': + package = 'f' * 64 + elif kind == 'policy': + changed['configs']['policy_file']['sha256'] = 'f' * 64 + changed['routing']['policy']['sha256'] = 'f' * 64 + else: + changed['task']['lead']['mode'] = 'headless' + peer = profile('lead', 'lead-model') + changed['routing']['roles']['lead'] = { + 'selected': {'profile_id': 'lead', 'profile': peer, + 'profile_sha256': digest(peer)}, 'fallbacks': [], + 'fallback_mode': 'none', + } + after = api.paired_input_identity(changed, role='reviewer', package_digest=package) + self.assertEqual(base['case_sha256'], after['case_sha256']) + self.assertNotEqual(base['input_sha256'], after['input_sha256']) + + def test_tested_role_fallback_policy_is_not_the_experimental_variable(self): + api = self.api() + before = frozen_snapshot() + base = api.paired_input_identity(before, role='reviewer', package_digest='e' * 64) + changed = copy.deepcopy(before) + changed['routing']['roles']['reviewer']['fallback_mode'] = 'policy' + after = api.paired_input_identity(changed, role='reviewer', package_digest='e' * 64) + self.assertEqual(base['case_sha256'], after['case_sha256']) + self.assertNotEqual(base['input_sha256'], after['input_sha256']) + + def test_task_check_and_candidate_changes_cannot_share_an_input_identity(self): + api = self.api() + before = frozen_snapshot() + base = api.paired_input_identity(before, role='reviewer', package_digest='e' * 64) + for kind in ('goal', 'check', 'candidate', 'scope'): + with self.subTest(kind=kind): + changed = copy.deepcopy(before) + if kind == 'goal': + changed['task']['goal'] = 'A different bounded task.' + elif kind == 'check': + changed['task']['checks'] = [{ + 'id': 'required', 'argv': ['true'], 'cwd': '.', + 'timeout_seconds': 1, 'required_to_pass': True, + }] + elif kind == 'candidate': + changed['workspace']['candidate_sha256'] = 'f' * 64 + else: + changed['task']['scope']['read_paths'] = ['different'] + after = api.paired_input_identity(changed, role='reviewer', package_digest='e' * 64) + self.assertNotEqual(base['case_sha256'], after['case_sha256']) + self.assertNotEqual(base['input_sha256'], after['input_sha256']) + + def test_missing_or_inconsistent_frozen_inputs_fail_closed(self): + api = self.api() + for kind in ('policy', 'role', 'profile', 'candidate', 'package'): + with self.subTest(kind=kind): + changed = frozen_snapshot() + package = 'e' * 64 + if kind == 'policy': + changed['routing']['policy']['sha256'] = '0' * 64 + elif kind == 'role': + changed['routing']['roles'] = {} + elif kind == 'profile': + changed['routing']['roles']['reviewer']['selected']['profile_sha256'] = '0' * 64 + elif kind == 'candidate': + del changed['workspace'] + else: + package = None + with self.assertRaises(ContractError): + api.paired_input_identity(changed, role='reviewer', package_digest=package) + + def test_missing_tested_native_adapter_fails_closed(self): + snapshot = frozen_snapshot() + selected = snapshot['routing']['roles']['reviewer']['selected'] + selected['profile']['harness'] = 'codex' + selected['profile_sha256'] = digest(selected['profile']) + with self.assertRaisesRegex(ContractError, 'adapter'): + self.api().paired_input_identity(snapshot, role='reviewer', package_digest='e' * 64) + + +if __name__ == '__main__': + unittest.main() diff --git a/test/core/test_experiment_reuse.py b/test/core/test_experiment_reuse.py new file mode 100644 index 0000000..53eb064 --- /dev/null +++ b/test/core/test_experiment_reuse.py @@ -0,0 +1,110 @@ +"""Reject experiment sample inflation through reused outcome observations.""" + +from __future__ import annotations + +from pathlib import Path +import subprocess +import tempfile +import unittest + +from test_learning import NOW, experiment, experimental_final + +from devsquad.contracts import ContractError +from devsquad.learning import evaluate_experiment, validate_experiment +from devsquad.store import Store + + +def repeated_pair_spec(project_path: Path) -> dict: + """The original regression: two observations relabelled as three pairs.""" + spec = experiment(project_path) + control_id = spec["cases"][0]["control_outcome_id"] + candidate_id = spec["cases"][0]["candidate_outcome_id"] + for case in spec["cases"]: + case["control_outcome_id"] = control_id + case["candidate_outcome_id"] = candidate_id + return spec + + +def outcome_chains(spec: dict) -> dict: + chains = {} + for case in spec["cases"]: + for arm, verdict in (("control", "failed"), ("candidate", "succeeded")): + outcome_id = case[f"{arm}_outcome_id"] + chains.setdefault(outcome_id, { + "final": experimental_final(outcome_id, verdict), + "late_corrections": [], + }) + return chains + + +class ExperimentObservationReuseTest(unittest.TestCase): + def test_validator_rejects_two_outcomes_relabelled_as_three_cases(self): + spec = repeated_pair_spec(Path("/tmp/experiment-reuse-project")) + self.assertEqual(len(outcome_chains(spec)), 2) + self.assertEqual(len(spec["cases"]), 3) + with self.assertRaises(ContractError): + validate_experiment(spec) + + def test_validator_rejects_cross_arm_reuse_in_different_cases(self): + spec = experiment(Path("/tmp/experiment-reuse-project")) + spec["cases"][1]["control_outcome_id"] = spec["cases"][0]["candidate_outcome_id"] + # Each local pair is distinct; uniqueness must cover all arms/cases. + self.assertTrue(all( + case["control_outcome_id"] != case["candidate_outcome_id"] + for case in spec["cases"] + )) + with self.assertRaises(ContractError): + validate_experiment(spec) + + def test_validator_rejects_same_arm_reuse_across_splits(self): + spec = experiment(Path("/tmp/experiment-reuse-project")) + spec["cases"][2]["candidate_outcome_id"] = spec["cases"][0]["candidate_outcome_id"] + self.assertNotEqual(spec["cases"][0]["split"], spec["cases"][2]["split"]) + with self.assertRaises(ContractError): + validate_experiment(spec) + + def test_evaluator_cannot_promote_two_observations_as_three_pairs(self): + spec = repeated_pair_spec(Path("/tmp/experiment-reuse-project")) + chains = outcome_chains(spec) + with self.assertRaises(ContractError): + evaluate_experiment(spec, chains, evaluated_at=NOW.isoformat()) + + def test_store_rejects_reused_outcomes_without_persisting_evaluation(self): + with tempfile.TemporaryDirectory(prefix="devsquad-experiment-reuse-") as temporary: + root = Path(temporary) + repo = root / "repo" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + store = Store(root / "state.sqlite3", root / "artifacts") + try: + spec = repeated_pair_spec(repo) + chains = outcome_chains(spec) + for outcome_id, chain in chains.items(): + final = chain["final"] + claim = store.claim_start(repo, f"run-{outcome_id}", {}, "owner") + # Seed the same legacy terminal fixture used by test_learning. + # It must never make reused observations valid evidence. + store.connection.execute( + "UPDATE runs SET state=?,phase=NULL WHERE id=?", + (final["verdict"], claim.run_id), + ) + store.record_outcome(claim.run_id, final, now=NOW) + rejected = None + try: + store.evaluate_learning_experiment(spec, now=NOW) + except ContractError as exc: + rejected = exc + self.assertEqual( + store.connection.execute( + "SELECT COUNT(*) FROM experiments WHERE experiment_id=?", + (spec["experiment_id"],), + ).fetchone()[0], + 0, + "invalid reused evidence must not leave a saved evaluation", + ) + self.assertIsInstance(rejected, ContractError) + finally: + store.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_experiment_v2_evaluation.py b/test/core/test_experiment_v2_evaluation.py new file mode 100644 index 0000000..51084ab --- /dev/null +++ b/test/core/test_experiment_v2_evaluation.py @@ -0,0 +1,212 @@ +"""Pure normalized-chain tests, not public or real-run provenance proof.""" + +from __future__ import annotations + +import copy +from datetime import timedelta +import unittest + +from test_experiment_provenance import digest, spec_v2 +from test_learning import NOW, experimental_final + +from devsquad.contracts import ContractError +from devsquad.experiment_provenance import assignment_for +from devsquad.learning import evaluate_experiment + + +PROJECT_COMMON_DIR = "/tmp/provenance-project/.git" + + +def normalized_chains(spec): + """Fabricate the normalized reader boundary, never saved-run authority.""" + chains = {} + for case in spec["cases"]: + for arm, verdict in (("control", "failed"), ("candidate", "succeeded")): + outcome_id = case[f"{arm}_outcome_id"] + final = experimental_final(outcome_id, verdict) + assignment = assignment_for( + spec, case["case_id"], arm, project_common_dir=PROJECT_COMMON_DIR, + ) + attempt_id = f"attempt-{outcome_id}" + chains[outcome_id] = { + "final": final, + "late_corrections": [], + "provenance": { + "run_id": f"run-{outcome_id}", + "assignment": assignment, + "profile_sha256": assignment["profile_sha256"], + "execution_sha256": assignment["execution_sha256"], + "input_sha256": assignment["input_sha256"], + "case_sha256": assignment["case_sha256"], + "attempt_ids": [attempt_id], + "attempts_sha256": digest([attempt_id]), + "final_outcome_sha256": digest(final), + "correction_sha256": [], + }, + } + return chains + + +def correction_for(final): + correction = experimental_final(f"escaped-{final['outcome_id']}", "succeeded") + correction.update({ + "kind": "late_correction", + "verdict": "escaped_defect", + "corrects_outcome_id": final["outcome_id"], + "observed_at": (NOW + timedelta(seconds=1)).isoformat(), + "summary": "A later check found an escaped defect in this exact arm.", + }) + return correction + + +class ExperimentV2EvaluationTest(unittest.TestCase): + def setUp(self): + self.spec = spec_v2() + self.chains = normalized_chains(self.spec) + + def evaluate(self, chains=None): + return evaluate_experiment( + self.spec, self.chains if chains is None else chains, + evaluated_at=(NOW + timedelta(seconds=2)).isoformat(), + project_common_dir=PROJECT_COMMON_DIR, + ) + + def test_disjoint_paired_observations_produce_a_bound_proposal(self): + result = self.evaluate() + self.assertEqual(result["schema_version"], 2) + self.assertEqual(result["verdict"], "promotion_proposal") + self.assertFalse(result["active_policy_changed"]) + self.assertEqual(result["metrics"]["evaluation"]["available_pairs"], 2) + self.assertEqual(result["metrics"]["held_out"]["available_pairs"], 1) + self.assertEqual(result["metrics"]["evaluation"]["control_successes"], 0) + self.assertEqual(result["evidence_sha256"], digest({ + outcome_id: chain["provenance"] + for outcome_id, chain in self.chains.items() + })) + self.assertEqual(self.evaluate(), result) + + def test_failed_candidate_remains_visible_and_does_not_qualify(self): + chain = self.chains["candidate-hold-1"] + chain["final"]["verdict"] = "failed" + chain["provenance"]["final_outcome_sha256"] = digest(chain["final"]) + result = self.evaluate() + self.assertEqual(result["verdict"], "no_change") + self.assertEqual(result["metrics"]["held_out"]["available_pairs"], 1) + self.assertEqual(result["metrics"]["held_out"]["candidate_successes"], 0) + self.assertIn({"case_id": "hold-1", "reason": "candidate_not_successful"}, result["failures"]) + + def test_missing_partner_remains_visible_without_counting_a_pair(self): + before = self.evaluate() + for arm in ("control", "candidate"): + with self.subTest(arm=arm): + chains = copy.deepcopy(self.chains) + missing_id = f"{arm}-hold-1" + del chains[missing_id] + result = self.evaluate(chains) + self.assertEqual(result["verdict"], "no_change") + self.assertEqual(result["metrics"]["held_out"]["available_pairs"], 0) + row = next(row for row in result["cases"] if row["case_id"] == "hold-1") + self.assertEqual(row["status"], "missing") + self.assertEqual(row["missing"], [arm]) + self.assertIn("insufficient_held_out_pairs", result["reasons"]) + self.assertNotEqual(result["evidence_sha256"], before["evidence_sha256"]) + + def test_duplicate_runs_are_rejected_even_with_a_missing_partner(self): + for missing_partner in (False, True): + with self.subTest(missing_partner=missing_partner): + chains = copy.deepcopy(self.chains) + chains["candidate-eval-2"]["provenance"]["run_id"] = chains["candidate-eval-1"]["provenance"]["run_id"] + if missing_partner: + del chains["control-eval-2"] + with self.assertRaisesRegex(ContractError, "run ids.*unique"): + self.evaluate(chains) + + def test_duplicate_attempts_are_rejected_even_with_a_missing_partner(self): + for missing_partner in (False, True): + with self.subTest(missing_partner=missing_partner): + chains = copy.deepcopy(self.chains) + prior = chains["candidate-eval-1"]["provenance"] + current = chains["candidate-eval-2"]["provenance"] + current["attempt_ids"] = list(prior["attempt_ids"]) + current["attempts_sha256"] = prior["attempts_sha256"] + if missing_partner: + del chains["control-eval-2"] + with self.assertRaisesRegex(ContractError, "attempt ids.*unique"): + self.evaluate(chains) + + def test_crossed_assignments_profiles_and_inputs_fail_closed(self): + changes = [ + (("assignment", "experiment_id"), "foreign-experiment"), + (("assignment", "spec_sha256"), "0" * 64), + (("assignment", "project_common_dir"), "/tmp/foreign-project/.git"), + (("assignment", "case_id"), "eval-2"), + (("assignment", "split"), "held_out"), + (("assignment", "arm"), "control"), + (("assignment", "role"), "implementer"), + (("assignment", "profile_id"), "profile-a"), + (("profile_sha256",), self.spec["variable"]["control_profile_sha256"]), + (("execution_sha256",), "0" * 64), + (("input_sha256",), self.spec["cases"][1]["input_sha256"]), + (("case_sha256",), self.spec["cases"][1]["case_sha256"]), + ] + for path, replacement in changes: + with self.subTest(path=path): + chains = copy.deepcopy(self.chains) + target = chains["candidate-eval-1"]["provenance"] + for field in path[:-1]: + target = target[field] + target[path[-1]] = replacement + with self.assertRaises(ContractError): + self.evaluate(chains) + + def test_final_outcome_bytes_must_match_the_saved_hash(self): + chains = copy.deepcopy(self.chains) + chains["candidate-eval-1"]["final"]["summary"] = "Changed after the witness was frozen." + with self.assertRaisesRegex(ContractError, "final outcome hash"): + self.evaluate(chains) + + def test_strict_late_correction_blocks_promotion_and_changes_evidence_digest(self): + before = self.evaluate() + chain = self.chains["candidate-hold-1"] + correction = correction_for(chain["final"]) + chain["late_corrections"] = [correction] + chain["provenance"]["correction_sha256"] = [digest(correction)] + after = self.evaluate() + self.assertEqual(after["verdict"], "no_change") + self.assertEqual(after["metrics"]["held_out"]["candidate_escaped_defects"], 1) + self.assertIn("candidate_escaped_defect_limit_exceeded", after["reasons"]) + self.assertIn({"case_id": "hold-1", "reason": "candidate_escaped_defect"}, after["failures"]) + self.assertNotEqual(after["evidence_sha256"], before["evidence_sha256"]) + + def test_malformed_or_crossed_late_corrections_are_rejected(self): + changes = [ + ("corrects_outcome_id", "candidate-eval-1"), + ("selection_mode", "automatic"), + ("observed_at", (NOW - timedelta(seconds=1)).isoformat()), + ("evidence_refs", []), + ("kind", "late"), + ] + for field, replacement in changes: + with self.subTest(field=field): + chains = copy.deepcopy(self.chains) + chain = chains["candidate-hold-1"] + correction = correction_for(chain["final"]) + correction[field] = replacement + chain["late_corrections"] = [correction] + chain["provenance"]["correction_sha256"] = [digest(correction)] + with self.assertRaises(ContractError): + self.evaluate(chains) + + def test_correction_hash_cannot_omit_or_misrepresent_saved_correction(self): + for invalid_hashes in ([], ["0" * 64]): + with self.subTest(hashes=invalid_hashes): + chains = copy.deepcopy(self.chains) + chain = chains["candidate-hold-1"] + chain["late_corrections"] = [correction_for(chain["final"])] + chain["provenance"]["correction_sha256"] = invalid_hashes + with self.assertRaises(ContractError): + self.evaluate(chains) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index fbbea6f..bef5fcb 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 13, + 14, ) tables = { row[0] @@ -684,7 +684,7 @@ def test_installed_wheel_applies_schema_four_to_twelve(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 13 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 14 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_learning.py b/test/core/test_learning.py index 4515837..c3573dd 100644 --- a/test/core/test_learning.py +++ b/test/core/test_learning.py @@ -429,7 +429,7 @@ def test_current_schema_contains_outcome_ledger(self): store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0], - 13, + 14, ) columns = { row[1] for row in store.connection.execute("PRAGMA table_info(outcomes)") diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index 5251a62..f356664 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -598,7 +598,7 @@ def test_schema_twelve_contains_lifecycle_ledger(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 13) + self.assertEqual(version, 14) tables = { row[0] for row in self.store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_store.py b/test/core/test_store.py index d63d7fe..ef4af0e 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -482,8 +482,8 @@ def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): ) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(14,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 14) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(15,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -498,7 +498,7 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 14) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") @@ -515,7 +515,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],13) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],14) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From b766f9e188dbb74a50812703ead4612eeba3e6eb Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 00:34:17 -0700 Subject: [PATCH 137/197] WIP checkpoint: record R3a verification and R3b next action (2026-10-01 00:34) --- docs/plans/engineering-team/M6-STATUS.md | 9 ++- docs/plans/engineering-team/RESUME.md | 17 ++++-- .../engineering-team/SOL-REVIEW-FOLLOWUP.md | 16 +++-- docs/plans/engineering-team/backlog.json | 2 +- .../R3a-provenance-contract-2026-10-01.json | 61 +++++++++++++++++++ 5 files changed, 91 insertions(+), 14 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R3a-provenance-contract-2026-10-01.json diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md index 08588f4..80cae05 100644 --- a/docs/plans/engineering-team/M6-STATUS.md +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -11,10 +11,13 @@ separate measurement; it does not block this engineering work. R3a now rejects reused outcomes and adds v2 concrete-profile/paired-input contracts, a stable corpus identity distinct from runtime context, strict normalized evidence chains and schema-14 prelaunch assignment persistence. -The focused 47-test learning/lifecycle/experiment gate passes. This is partial: +The focused 47-test gate and complete 402-test gate (two optional-SDK skips) +pass at `672e383`, with bounded independent follow-up and 227 Bash assertions. +The SQLite cleanup warning remains R5. This is partial: the saved-run v2 reader, shared lifecycle eligibility, historical compatibility -audit and public trial execution are not complete. See RESUME for the current -review/full-gate status; do not treat legacy helper fixtures as public proof. +audit and public trial execution are not complete. See +[R3a evidence](evidence/R3a-provenance-contract-2026-10-01.json); +do not treat legacy helper fixtures as public proof. | Requirement | Planned evidence | Status | |---|---|---| diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 64886ca..bd36413 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,7 +6,7 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Review correction and next action -R3a is now implemented after `b3de5c6`: global outcome-reuse rejection, +R3a is source/offline verified at `672e383`: global outcome-reuse rejection, versioned profile/input/assignment contracts, separate corpus-versus-pair fingerprints, schema-14 prelaunch assignment persistence under the preparation fence, and strict normalized v2 chain evaluation. The original five reuse @@ -15,7 +15,13 @@ learning and lifecycle gate currently passes 47 tests; nine targeted migration checks, including installed-wheel upgrades, also passed. Bounded review found omitted tested-role fallback policy and native execution identity; both now have regressions and fixes, including predeclared per-arm execution hashes. -Independent follow-up and the full integration gate remain pending here. +The independent two-test follow-up passed with no remaining concrete finding +in the fixes. The full unchanged-source gate ran **402 tests in 213.208 seconds, +OK with two optional-SDK skips**; UTC/monotonic wrapper timing agreed at +213.376 seconds. The known SQLite finalizer warning remains R5. All 227 Bash +assertions and the generated-reference check passed. See +[R3a evidence](evidence/R3a-provenance-contract-2026-10-01.json). No test is +still running; begin R3b without repeating this unchanged gate. **R3 remains in progress.** The v2 saved-run outcome reader is not wired into `Store.evaluate_learning_experiment` yet, so public v2 evaluation fails closed @@ -366,8 +372,11 @@ provider paths must not be advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Complete R3a's pending independent review/integration gate, then implement - the R3b saved-run v2 outcome reader and current-evidence eligibility. Finish +2. Implement the R3b saved-run v2 outcome reader and current-evidence + eligibility. Join `experiment_specs` / `experiment_assignments` to actual + role attempts and final/correction outcome rows; recompute original input + and per-arm execution fingerprints from the immutable preparation witness. + Validate every relevant attempt, including fallbacks/repairs. Finish R3b/R3c saved evidence, replay, qualification/promotion/rollback/catalog-fallback eligibility and historical compatibility, then execute R4–R6 in dependency order. Preserve R1/R2, diff --git a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md index e2bb65b..dccbef8 100644 --- a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md +++ b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md @@ -29,10 +29,13 @@ open under R5; this is not a warning-free or installed/live pass. See [R2 evidence](evidence/R2-observed-identity-2026-10-01.json). Start at **R3**. R3–R8, C1 and affected installed/live proofs remain open. -R3a continuation after `b3de5c6`: outcome uniqueness, the v2 provenance +R3a source/offline checkpoint at `672e383`: outcome uniqueness, the v2 provenance contract, separate corpus/pair hashes, schema-14 fenced prelaunch assignments -and normalized-chain checks are implemented. Focused tests pass; full gate -and independent patch review are pending. The saved-run v2 outcome reader and +and normalized-chain checks are verified by 47 focused tests, migration checks, +the 402-test complete gate (two optional-SDK skips) and bounded independent +review/follow-up. The SQLite warning remains R5. See +[R3a evidence](evidence/R3a-provenance-contract-2026-10-01.json). +The saved-run v2 outcome reader and shared lifecycle eligibility are still R3b; do not describe this partial checkpoint as trustworthy end-to-end qualification or a public trial runner. @@ -50,8 +53,9 @@ CRLF native output retains its exact byte hash. The original Preserve the verification history: a prior 367-test run failed six cases; all six passed unchanged with stable clocks, and the final full run passed with UTC and monotonic elapsed times agreeing. Do not erase the failed run or -weaken budgets to make tests pass. The current request is a planning handoff: -no additional implementation, installation or provider call was performed. +weaken budgets to make tests pass. The planning handoff at `b3de5c6` made no +implementation, installation or provider call; subsequent R3a source work is +recorded above. No installation or provider call was needed for that slice. | Next slice | Deliverable | Gate before claiming completion | |---|---|---| @@ -465,7 +469,7 @@ permission to claim routing improvement. ## First action for Sol Read the recovery files and verify current Git state. Preserve `ef98889` and -later work. Finish R3a's pending review/full gate, then implement **R3b**'s +later work. R3a's review and full gate are verified; implement **R3b**'s saved-run evidence reader and shared current-evidence gate. Complete R3c before R4. Preserve R1's check-integrity gate and R2's actual-attempt/native-identity gate. Continue R3–R6 without waiting on the Claude login or Jev key. Retain R7/R8 diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index e90ba35..ca5d0c4 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -24,7 +24,7 @@ "review_work_packages": [ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, - {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "checkpoint": "R3a outcome uniqueness, v2 profile/execution/input/assignment contracts, schema-14 fenced prelaunch persistence and normalized-chain evaluator implemented after b3de5c6; 47 focused and nine migration tests pass, full gate/review follow-up pending. Reviewed fallback/native-context gaps have regressions and fixes. Public v2 saved-run reader and shared eligibility remain R3b; legacy/public/correction proof remains R3c. R3 is not closed."}, + {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3a-provenance-contract-2026-10-01.json", "checkpoint": "R3a source/offline verified at 672e383: outcome uniqueness, v2 profile/execution/input/assignment contracts, schema-14 fenced prelaunch persistence and normalized-chain evaluator; 47 focused, nine migration checks and complete 402-test gate pass with two optional-SDK skips. Review findings repaired and independently rechecked; SQLite warning remains R5. Public v2 saved-run reader and shared eligibility remain R3b; legacy/public/correction proof remains R3c. R3 is not closed."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, {"id": "R6", "title": "Normal terminal experience and readiness", "status": "pending", "milestones": ["M5", "M7"], "depends_on": ["R1", "R2", "R4", "R5"], "items": ["G3", "G4"]}, diff --git a/docs/plans/engineering-team/evidence/R3a-provenance-contract-2026-10-01.json b/docs/plans/engineering-team/evidence/R3a-provenance-contract-2026-10-01.json new file mode 100644 index 0000000..46911f2 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3a-provenance-contract-2026-10-01.json @@ -0,0 +1,61 @@ +{ + "schema_version": 1, + "work_package": "R3a", + "status": "partial_source_verified", + "implementation_revision": "672e383", + "recorded_on": "2026-10-01", + "scope": "Outcome uniqueness, v2 provenance contracts, schema-14 fenced prelaunch assignments and pure normalized-chain evaluation; not end-to-end R3 closure", + "baseline_reproductions": [ + "Five new reused-outcome tests failed before production changes: repeated pair, cross-arm reuse, cross-split reuse, evaluator acceptance and invalid persisted evaluation. All five now pass.", + "The first v2 contract run had one failure and seven errors across nine tests because the new schema/module did not exist; this is a missing-feature baseline, not eight reproduced production exploits.", + "A subsequent regression confirmed that unbound v2 outcome labels could still be evaluated after schema support; requiring saved-project and normalized provenance now rejects them.", + "An independent normalized-chain test exposed the incorrect late-correction kind check; the strict valid correction now changes the evidence digest and blocks the proposal.", + "Review exposed omitted tested-role fallback policy and native execution context. Dedicated regressions failed before the fixes; fallback settings now affect pair context and each arm explicitly predeclares its profile/native execution fingerprint." + ], + "acceptance_mapping": [ + {"requirement": "No global outcome reuse across arms/cases/splits or invalid new evaluation persistence", "tests": "test/core/test_experiment_reuse.py: 5 regressions"}, + {"requirement": "Strict v2 profile/execution/assignment contracts and distinct corpus versus controlled-pair fingerprints", "tests": "test/core/test_experiment_provenance.py: 12 contract regressions"}, + {"requirement": "Assignments precede attempts under the preparation fence; rebinding, duplicate arms, changed specs and invalid input fail atomically", "tests": "test/core/test_experiment_assignment_store.py: 8 persistence/migration regressions"}, + {"requirement": "Normalized chains reject reused runs/attempts and crossed profiles/inputs even with missing partners; failures, missingness and corrections remain visible", "tests": "test/core/test_experiment_v2_evaluation.py: 10 pure evaluator regressions"} + ], + "verification": { + "focused_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_experiment_reuse test_experiment_provenance test_experiment_assignment_store test_experiment_v2_evaluation test_learning test_lifecycle", + "focused_result": "47 tests passed in 1.470 seconds after the final source change", + "migration_checks": "Nine independently run targeted migration tests passed, including both installed-wheel upgrade checks; existing schema-version expectations were updated without changing unrelated run-version fixtures", + "independent_review": "Bounded R3a audit found tested native execution metadata was not bound. The repaired predeclared per-arm execution hash passed an independent two-test follow-up in 0.098 seconds, with no remaining concrete finding in those changes. The deferred reader/lifecycle/public-controller paths were outside this audit.", + "bash": "11 files and 227 assertions passed before the source checkpoint", + "generated_reference": "python3 scripts/generate-core-reference.py --check: current", + "core": { + "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src python3 -m unittest discover -s test/core -v", + "result": "OK (skipped=2)", + "tests_run": 402, + "failures": 0, + "errors": 0, + "skipped": 2, + "suite_seconds": 213.208, + "wrapper_wall_seconds": 213.376, + "wrapper_monotonic_seconds": 213.376, + "exit_code": 0, + "notes": "Source remained fixed at 672e383. Two official MCP SDK tests skipped because that optional dependency is absent in this interpreter. An ignored finalizer ResourceWarning for an unclosed SQLite connection appeared during a cross-process test; it remains R5 and the suite is not warning-free." + } + }, + "source_sha256": { + "plugin/core/src/devsquad/learning.py": "4558b53cc97ac1c00a1a6f3480f757859dc4fa6164ba3a7a257658050393f8b8", + "plugin/core/src/devsquad/experiment_provenance.py": "c0e98d3371679f50ddaa0652092382b434bac46bca6a9fdb6e5b4bace34a1934", + "plugin/core/src/devsquad/store.py": "362b815eb818054c013c33e13573595222acf4b48ff62c77ee82a72e13e8345e", + "plugin/core/src/devsquad/migrations/014_experiment_assignments.sql": "e0675f9c71df3ded92467835972792a90f78ae17a4a86e5a88335eb90262afa7" + }, + "test_sha256": { + "test/core/test_experiment_reuse.py": "8f1b9916867c0192de3d10b8ea7231a44261766b93407d2f5ae20039dbdc0dc0", + "test/core/test_experiment_provenance.py": "eece72a69ae68dda7caf20205864b0deef188307ddfca8911cfcf38601f21c91", + "test/core/test_experiment_assignment_store.py": "ca1d39f1f725e5f45eabcb9ac9d0890c09f184e6e71cba372c4891df8ff7e3ab", + "test/core/test_experiment_v2_evaluation.py": "5f1da4d399082ce6510a60126d3095481ae1d4d22718b6cb18a75b96deb5046a" + }, + "remaining_requirements": [ + "R3b must build v2 chains from immutable assignment records, actual saved attempts and outcome/correction rows. Store's current public evaluation reader does not yet supply that provenance, so v2 evaluation fails closed.", + "R3b must share current-evidence eligibility across evaluation/qualification replay, new qualification, promotion, regression rollback and catalog fallback. Existing v1 label-based positives are not proof of repaired eligibility.", + "R3c must preserve readable legacy audit/proposal evidence without authorizing new decisions; schema migration preserving bytes alone does not complete that requirement.", + "Public end-to-end realistic experiment/lifecycle fixtures and the R5 paired-trial controller are still required. Pure normalized witnesses in this artifact are deliberately fabricated unit inputs, not saved-run authority or live model proof.", + "The full R1-R8/M1-M7+C1 objective remains active. No provider call, installed runtime refresh, paid API fallback or global setting change occurred." + ] +} From a22b9498a1ba56b7020303d87bef8032df29b27e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 00:43:57 -0700 Subject: [PATCH 138/197] WIP checkpoint: docs: give SOL current review feedback and ordered acceptance plan (2026-10-01 00:43) --- docs/plans/engineering-team/RESUME.md | 21 +++ .../engineering-team/SOL-REVIEW-FOLLOWUP.md | 161 ++++++++++++++++-- docs/plans/engineering-team/backlog.json | 11 ++ 3 files changed, 176 insertions(+), 17 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index bd36413..3764dd6 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,27 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Review correction and next action +Latest requested action was a review and execution plan for SOL. The October 1 +planning pass inspected `b766f9e` / source `672e383` and updated +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), without runtime changes, +installation or provider calls. Start with **R3b.1** (saved-run reader), then +**R3b.2** (shared eligibility and explicit evaluation/review revisions), then +R3c's realistic public and historical-compatibility proof. The plan now maps +each lifecycle consumer to its acceptance gate, clarifies R5 trial entrypoints +and terminal projection coverage, and requires an active-run schema-upgrade +safety test (coexistence or explicit deferral) before R8 refreshes the +installation. These are next-work requirements, not newly completed gates. +The verified source baseline below is unchanged; do not rerun it merely +because the planning files changed. + +Planning verification: all eight R3a source/test SHA256 values still match the +saved evidence; backlog JSON, generated-reference and whitespace checks pass. +The initial Bash run was interrupted after sandbox-denied process inspection +caused three portable-timeout/cleanup assertions to fail; it is not a pass. +Its verified test tree was stopped (exit 137). The single rerun with required +process access passed all **227 assertions in 11 files**. No test remains live. +The full Python gate was not repeated for documentation-only changes. + R3a is source/offline verified at `672e383`: global outcome-reuse rejection, versioned profile/input/assignment contracts, separate corpus-versus-pair fingerprints, schema-14 prelaunch assignment persistence under the preparation diff --git a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md index dccbef8..8916716 100644 --- a/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md +++ b/docs/plans/engineering-team/SOL-REVIEW-FOLLOWUP.md @@ -7,8 +7,8 @@ review findings below, finish the missing runtime connections, and prove the requested product through the public service and installed normal commands. This is a continuation of [SOL-HANDOFF.md](SOL-HANDOFF.md), not a redesign. The full delivery remains M1–M7 plus C1; the immediate engineering priority is -the shared acceptance repair followed by M5–M7. C1 is optional to invoke but -required to implement under the original assignment. +saved-run evidence and public runtime integration in R3–R6. C1 is optional to +invoke but required to implement under the original assignment. Review baseline: `f4fa6577e2e891151231c9c7d3180be6e9e23faa`. Compare Git state before acting and retain later work. The review made no source changes. @@ -20,14 +20,14 @@ Read `AGENTS.md`, `CONTRIBUTING.md`, [RESUME.md](RESUME.md), this document and for each work package. Do not reload the historical chat or restart M1/M2. This review supersedes the earlier claim that only credentials remain. -Current execution checkpoint: **R1 and R2 source/offline repairs verified**. -The implementation baseline is `ef988897e2dda495482dd26e7c583bc428086ba0`; -preserve any later commits. R2's final complete run passed 367 tests with two -optional-SDK skips in 215.886 seconds. The SQLite finalizer warning remains -open under R5; this is not a warning-free or installed/live pass. See -[R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json) and -[R2 evidence](evidence/R2-observed-identity-2026-10-01.json). Start at **R3**. -R3–R8, C1 and affected installed/live proofs remain open. +Current inspected checkpoint: **`b766f9e` on `codex/engineering-team`**, with a +clean tree before this planning update. The latest implementation is +`672e383`; preserve it and all later work. **R1, R2 and R3a are source/offline +verified; start at R3b, not R1.** R3 as a whole, R4–R8, C1 and affected +installed/live proofs remain open. See +[R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json), +[R2 evidence](evidence/R2-observed-identity-2026-10-01.json) and the R3a record +below. The installed release named above predates these repairs. R3a source/offline checkpoint at `672e383`: outcome uniqueness, the v2 provenance contract, separate corpus/pair hashes, schema-14 fenced prelaunch assignments @@ -39,6 +39,12 @@ The saved-run v2 outcome reader and shared lifecycle eligibility are still R3b; do not describe this partial checkpoint as trustworthy end-to-end qualification or a public trial runner. +This October 1 follow-up is a **source review and planning handoff**, not +another implementation or live verification. The recorded 402-test result +belongs to `672e383`, not a new run for this document. Two optional-SDK skips +and the SQLite finalizer warning remain explicit. No installation, login, +provider request or model-setting change is part of this planning pass. + ### Ready-to-execute handoff for Sol Do not reimplement R2. It now has a shared strict native parser, v2 imported @@ -60,7 +66,7 @@ recorded above. No installation or provider call was needed for that slice. | Next slice | Deliverable | Gate before claiming completion | |---|---|---| | R1 / R2 | Preserve verified source repairs | Recheck affected regressions when shared code changes; installed/live proof remains R8 | -| R3a | Strict experiment provenance contract | Reused outcomes/runs/splits and mismatched paired inputs rejected | +| R3a — verified | Preserve provenance contracts and fenced assignments | Existing contract/store tests cover this foundation; normalized unit witnesses do not prove actual saved-run evidence | | R3b | Durable validation and current-evidence eligibility | Replay, qualification, promotion, rollback and catalog fallback cannot reuse stale/unverified evidence | | R3c | Public regression and migration evidence | Valid independent pairs work; legacy receipts remain readable; late corrections block new unsafe decisions | | R4 → R5 → R6 | Routing, learning and normal terminal integration | Public-run evidence through the full chain, then fresh-install usability proof | @@ -79,7 +85,7 @@ Preserve the substantial working implementation: the durable runner and ledger, process ownership and recovery, isolated worktrees, fenced host handoffs, Codex protocol repairs, immutable installer and MCP integrations. The successful installed Codex review receipt is real and remains valid for -that exact run. Its SHA256 was rechecked: +that exact run. Its SHA256 was rechecked in the September 29 review: `0103db19a528cff916ebef80c6ec94b682ed4801fc22db020cdbc21796040f70`. The main problem is the strength of completion claims. Several components @@ -113,6 +119,11 @@ Source locations refer to the baseline above and will move as repairs land. | G4 | Verification coverage / M5, M7 | Default check detection on DevSquad selects only `bash test/run.sh`; the Python core suite is omitted, including when the requested fix changes Python core code. Detection also inspects the current checkout rather than the selected target tree. | R6 | | G5 | Remaining full-delivery scope / C1 | `squad council` is absent. Its implementation and acceptance are included in the full assignment. | R7 | +The table records the original review baseline. G1's missing internal writer +is now partly addressed by R3a: `Store.complete_preparation` freezes schema-14 +assignments before attempts. The missing **public controller** and automatic +terminal-outcome projection remain R5; reuse the new writer, do not rebuild it. + The optional decision helper currently executes only a fake adapter; non-fixture requests record `unavailable`. This is consistent with the narrow M6-D1 contract, not proof of a production Jev integration. The one-request @@ -306,6 +317,89 @@ attempt profile indices. Do not trust caller-supplied `experimental` labels as proof. Failed and missing evidence must remain visible and ineligible for unearned success credit. Keep schema/contract changes and migrations together. +#### October 1 feedback: exact R3b/R3c implementation boundary + +The following gaps were rechecked in source at `672e383` / checkpoint +`b766f9e`. They are unfinished integration identified by this review, not new +claims of live exploitation or a reason to repeat the completed R3a suite. +Paths below are under `plugin/core/src/devsquad/` unless stated otherwise. + +| Current path | Concrete remaining gap | Required change | +|---|---|---| +| `store.py:1763`, `evaluate_learning_experiment` | Loads only outcome payloads; passes neither saved project identity nor normalized provenance required by v2 | Construct chains from immutable assignments and actual saved attempts; never accept a caller's witness as authority | +| `store.py:1780` and `store.py:2073` | Evaluation and qualification replay return before current-evidence checks | Separate immutable historical receipt from present eligibility; replay must not restore eligibility lost to corrections | +| `store.py:2114`, `record_profile_qualification` | Matches candidate by profile ID and alias, without binding the full tested profile/execution evidence | Check exact candidate fingerprint and the declared role/task context against validated saved evidence | +| `store.py:2248`, `store.py:2309`, `store.py:2532` | Promotion, regression rollback and catalog fallback rely on saved verdicts/qualification labels | Use the same current-evidence gate for every new decision, within the mutation transaction | +| `learning.py:690`, `build_learning_proposal` | Strict new-spec validation also parses historical v1 records; an old reused-outcome record can fail before a readable audit result | Preserve original bytes/hashes and report explicit historical-ineligible status without allowing new decisions | +| `test/core/test_learning.py:226`, `test/core/test_lifecycle.py:198` | Existing positive fixtures set terminal state directly with SQL | Replace positive authority proofs with distinct prepared runs, actual fixture attempts and public completion; retain SQL only for explicitly labelled migration/tamper negatives | + +**R3b.1 — authoritative saved-run reader.** Keep `learning.py`'s evaluator +pure and make Store responsible for constructing its trusted inputs. In one +consistent transaction, verify saved spec/assignment hashes and project +identity, recompute the original pair/corpus/execution fingerprints from +`experiment_assignments.snapshot_json` and its package digest, and bind every +relevant role attempt to that declaration. Include failed attempts, retries, +fallbacks and repairs, not only the eventual successful worker. An actual +different-profile fallback cannot be credited as the declared arm. Reconcile +durable attempt profile/index/package and frozen adapter identity; retain +native observed evidence without manufacturing missing identity. Preparation +must precede attempts by the existing fence/order, not timestamp comparison. +No attempt or missing partner is missing evidence, never an invented exposure. +Verify and hash final outcomes plus ordered corrections from saved rows. + +First acceptance: a minimum qualifying set of disjoint public offline runs +can be evaluated through Service/Store without supplied provenance; changing +the saved actual profile, assignment, package or paired input fails atomically, +including when the other arm is missing. A test-only preflight seam may attach +the predeclared manifest until R5 supplies the public controller, but it must +still use fenced preparation, real fixture attempts and public disposition. +Document that seam: it does not prove normal public experiment entry exists. + +**R3b.2 — one eligibility gate and explicit revisions.** Define one reusable +current-evidence result with concrete ineligibility reasons, bound to spec, +evaluation and evidence hashes. Keep historical decoding separate. Protect +new qualification and binding mutations against a correction arriving between +validation and commit; do not run a read check and then write from stale data. +Qualified profile fingerprints must match the tested candidate, with role and +task-class applicability checked rather than inferred from the alias alone. + +Settle the correction/re-evaluation contract before implementation: +`migrations/011_experiments.sql` permits only one evaluation per experiment ID, +and preparation already freezes that spec. A correction therefore cannot be +handled by silently overwriting its evaluation or repeatedly replaying the +same stale receipt. Use an explicit append-only evaluation/review revision +referencing the original spec and evidence, with qualifications/decisions +pinning the reviewed revision/hash. Update contract, migration and public +response semantics together. Do not reassign old runs to a new experiment to +make corrected evidence appear to be new independent samples. + +| Consumer | Acceptance for current evidence | Acceptance after a late correction / for unverified v1 history | +|---|---|---| +| Evaluation and its replay | Saved-run-derived evidence and deterministic hashes | Preserve receipt; expose stale/ineligible status or a specific error, never silently re-authorize it | +| Qualification and its replay | Exact candidate/context, measured pairs and current evidence match | Cannot return an unqualifiedly current `qualified` result; require explicit review/revision | +| New promotion | Validated qualification plus current evidence, in the same write transaction | Block without incrementing binding version | +| New regression rollback | Validate both regression evidence and any qualification-backed target | Stale/legacy regression or target cannot authorize the change | +| New catalog fallback | Select only a currently eligible, available predecessor | Skip stale qualification-backed predecessors; block if none remain | +| Report / proposal / historical audit | Show evidence scope and current eligibility | Remain readable; no promotion proposal from historical-ineligible evidence | +| Exact completed binding-decision replay | Return the original receipt without another mutation | Still return historical receipt; do not retroactively undo the completed decision | + +Preserve the existing explicitly proven bootstrap-baseline contract for a +predecessor without a qualification record; do not invent an experiment for +it or treat any arbitrary legacy profile as proven. A new correction blocks +future unsafe use; it does not itself authorize automatic binding changes. + +**R3c — public and upgrade proof, then close R3.** Cover branch-review and +issue-delivery input identities, honest failed/missing arms, correction after +evaluation and after qualification, valid promotion/rollback, rejected stale +catalog fallback, and repeated completion with no duplicate state mutation. +Upgrade a historical database and exercise actual report/proposal/read paths, +including previously saved duplicate-outcome evidence, not only SQL byte +preservation. Positive fixtures must meet the original sample/budget gates; +do not lower those gates or replace a valid-success test with rejection to +make the suite green. Run focused tests per change, one full integration gate +at the coherent R3 boundary, then a bounded independent review. Record R3 +closure separately from R5 controller and R8 installed/live acceptance. + ### R4 — Connect normal routing, catalog lifecycle and quota **Files:** `task_entry.py`, `catalog.py`, `capacity.py`, `router.py`, @@ -334,6 +428,9 @@ partial/auth-failed discovery does not remove models; two projects honor fresh weekly exhaustion despite short-window availability; stale/unsupported quota remains unknown; no permission or billing expansion occurs. Exercise these through normal task entry, not only router/store helper calls. +Include concurrent refreshes (one lease owner), interrupted refresh, and +account/config/version switches: incompatible scopes cannot reuse each +other's catalog or quota, and failed refresh preserves the last-good incumbent. The existing normal profiles are already labelled `trial`; the defect is bypassing stable alias/qualification policy, not a missing trial label. @@ -347,7 +444,8 @@ bypassing stable alias/qualification policy, not a missing trial label. Keep subjective later corrections explicit. Recover safely from a crash between terminalization and outcome projection without duplicate records. 2. Freeze the experiment, cases, splits, profiles and budgets before either - arm launches, and run bounded paired + arm launches. Reuse R3a's schema-14 assignment writer and R3b's saved-run + reader/eligibility; do not introduce another assignment authority. Run bounded paired trials through the existing runner, account reservations and experiment budgets. Derive consumed budget from durable attempts, not imported counters. Carry assignment, case/split and profile provenance into outcomes. Use a @@ -365,6 +463,19 @@ Exercise the complete public chain: freeze experiment → run both arms → reco terminal outcomes → evaluate → qualify → promote for a new run → rerun held-out cases → roll back. Count independent completed pairs, not labels or retries. +Use branch-review runs over one already-frozen candidate for reviewer trials, +and issue-delivery runs over one baseline/task/check contract for implementer +trials. Delivery preflight has no produced review candidate yet; assigning two +whole delivery runs as a controlled reviewer comparison would not establish +the same candidate. Assignment must still precede every attempt. + +Test outcome projection from each terminal origin: preparation failure/cancel, +worker failure, host completion, headless completion, and crash/replay around +the projection boundary. All yield exactly one objective final outcome, with +missing attempt/criteria evidence explicitly absent. A prelaunch failure must +not become a completed experiment exposure. Do not project success merely +because one successful terminal path passes. + ### R6 — Finish the normal terminal and readiness experience **Files:** `cli.py`, `task_entry.py`, `diagnostics.py`, integrations, runtime @@ -421,6 +532,20 @@ Inconclusive comparison is recorded as inconclusive, with automatic use off. dependency/drift/idempotence checks and required installed-SDK tests. Independently review the repaired paths. Preserve previous releases and saved receipts; a new release does not erase earlier failures. + **Before updating the real installation**, prove safe upgrade behavior + with active or recoverable old-package runs in a temporary install: + compatible coexistence across a real schema migration, or explicit upgrade + deferral until those runs are safely reconciled. Test continuation, cancel + and recovery for the selected approach. Frozen daemons load their old package + (`service.py:963`, `detached.py:84`), while `Store.migrate` rejects a newer + schema (`store.py:175`). The existing installer survival fixture changes + package versions, not schema (`test/core/test_install_core.py:58`). This is + a source-backed compatibility risk and missing test, not a reproduced live + failure. Cover schema 13 → 14 and any R3b evaluation-revision migration; + implement safe compatible coexistence or an explicit non-destructive + deferred upgrade, without stranding active runs or relaxing schema checks. + Deferral is a documented safety state, not proof of cross-schema + coexistence or completion of the installed-upgrade gate. 2. After normal Claude login, prove the M4 real-host handoff and one bounded installed Claude implementation → verified different-model Codex review → mandatory checks → disposition. After Grok login, prove its supported @@ -468,15 +593,17 @@ permission to claim routing improvement. ## First action for Sol -Read the recovery files and verify current Git state. Preserve `ef98889` and -later work. R3a's review and full gate are verified; implement **R3b**'s -saved-run evidence reader and shared current-evidence gate. Complete R3c before R4. +Read the recovery files and verify current Git state. Preserve `b766f9e`, its +`672e383` source checkpoint and later work. R3a's review and full gate are +verified; implement **R3b.1**'s authoritative saved-run reader first, then +**R3b.2**'s shared current-evidence gate and explicit evaluation/review revisions. +Use the consumer acceptance matrix above. Complete R3c before R4. Preserve R1's check-integrity gate and R2's actual-attempt/native-identity gate. Continue R3–R6 without waiting on the Claude login or Jev key. Retain R7/R8 and C1 in the full scope. Report each slice as red baseline, verified offline, installed proof, or externally blocked; never collapse those into one claim. -Suggested instruction to SOL: “Execute this plan from R3 on +Suggested instruction to SOL: “Execute this plan from R3b.1 on `codex/engineering-team`, in small spec → failing public test → implementation → verification → review → checkpoint cycles. Do not redo R1/R2 or reload the old chat. After each slice report the evidence, open gates and exact next diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index ca5d0c4..8a2f097 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -14,6 +14,17 @@ "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", "next_milestone": "M5", + "next_work_package": "R3b.1", + "planning_checkpoint": { + "recorded_on": "2026-10-01", + "inspected_checkpoint": "b766f9e", + "implementation_revision": "672e383", + "artifact": "SOL-REVIEW-FOLLOWUP.md", + "scope": "Source review and planning only; no runtime edits, installation or provider requests", + "next_action": "R3b.1 authoritative saved-run reader, then R3b.2 current-evidence eligibility and explicit evaluation/review revisions, then R3c public and historical compatibility proof", + "additional_acceptance": "Reuse R3a assignment authority in R5; cover all terminal outcome origins; prove old-package active-run schema-upgrade safety through compatible coexistence or explicit deferral before real R8 installation", + "verification": "Source/test hashes match R3a evidence; JSON, generated reference and diff checks pass. Restricted Bash run was interrupted after process-inspection-related cancellation failures; permitted rerun passed 227 assertions in 11 files. No full Python rerun for documentation-only changes." + }, "review_checkpoint": { "reviewed_revision": "f4fa6577e2e891151231c9c7d3180be6e9e23faa", "recorded_on": "2026-09-29", From 46e4230e44ade3a75d9965c3f5a87dfe1aa34fb4 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 12:59:38 -0700 Subject: [PATCH 139/197] WIP checkpoint: R3b.1 partial reader, focused verification and explicit failure-path next action (2026-10-01 12:59) --- docs/plans/engineering-team/RESUME.md | 56 ++- docs/plans/engineering-team/backlog.json | 11 +- .../R3b1-reader-partial-2026-10-01.json | 61 ++++ .../core/src/devsquad/experiment_evidence.py | 344 ++++++++++++++++++ plugin/core/src/devsquad/store.py | 102 ++++-- test/core/experiment_runtime_fixture.py | 209 +++++++++++ .../test_experiment_evidence_integrity.py | 180 +++++++++ test/core/test_experiment_saved_runs.py | 230 ++++++++++++ 8 files changed, 1152 insertions(+), 41 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R3b1-reader-partial-2026-10-01.json create mode 100644 plugin/core/src/devsquad/experiment_evidence.py create mode 100644 test/core/experiment_runtime_fixture.py create mode 100644 test/core/test_experiment_evidence_integrity.py create mode 100644 test/core/test_experiment_saved_runs.py diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 3764dd6..0018ae4 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,7 +6,42 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Review correction and next action -Latest requested action was a review and execution plan for SOL. The October 1 +Interrupted R3b.1 work newer than the planning checkpoint is now preserved as +an explicitly **partial** saved-run reader. `experiment_evidence.py` joins +immutable assignments, preparation fences, actual attempts, stream/artifact +hashes and outcomes; `Store.evaluate_learning_experiment` uses it for v2 and +rejects changed-evidence replay without rewriting the old receipt. +`Store.reserve_attempt` checks assigned trials against the frozen controlled +input before launch. The public fixture uses real offline workers and host +dispositions, with a test-only predeclared-assignment preparation seam; it is +not a production paired-trial controller or native model-quality proof. + +The imported-profile regression first failed (`ContractError not raised`) +and passed after semantic profile binding was added. Earlier focused tests +passed 59 cases in 43.570 seconds; the additional prelaunch-mutation test +passed separately. The current combined gate and checkpoint details are in +[R3b.1 partial evidence](evidence/R3b1-reader-partial-2026-10-01.json). +No full Python integration gate, installed refresh or provider call has run +for this reader slice. The independent follow-up review hit its usage limit; +do not claim an independent reader audit passed. + +**Exact next action:** reproduce the reader's handling of an honest terminal +failed reviewer whose output metadata has no `failure` key. The present +successful-review stdout decoder is guarded by that key and may incorrectly +decode failed non-JSON output. This is an open inspection concern, not yet a +reproduced/fixed regression. Resolve it and run the complete core gate before +accepting R3b.1. Then finish R3b.2's shared current-evidence eligibility and +explicit append-only evaluation/review revisions; legacy/public compatibility +remains R3c and the production paired-trial controller remains R5. + +Gemini/Antigravity's September 29 live `squad_status` receipt proves read-only +MCP observation of the same terminal-created saved run. It does not prove +Gemini implementation/review worker execution or the updated installation. +No extra Gemini login action is currently recorded; keep the normal session +signed in. Claude requires normal login; Grok's last verified authentication +was expired. The optional Jev key is separate from normal operation. + +The earlier requested action was a review and execution plan for SOL. The October 1 planning pass inspected `b766f9e` / source `672e383` and updated [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), without runtime changes, installation or provider calls. Start with **R3b.1** (saved-run reader), then @@ -44,11 +79,10 @@ assertions and the generated-reference check passed. See [R3a evidence](evidence/R3a-provenance-contract-2026-10-01.json). No test is still running; begin R3b without repeating this unchanged gate. -**R3 remains in progress.** The v2 saved-run outcome reader is not wired into -`Store.evaluate_learning_experiment` yet, so public v2 evaluation fails closed -rather than accepting outcome labels as provenance. R3b must join immutable -assignments, actual attempts and outcomes, then enforce one current-evidence -gate for replay/qualification/promotion/rollback/catalog fallback. Audit/read +**R3 remains in progress.** The partial reader above now connects v2 saved-run +evaluation, but its complete integration gate and failure-path audit remain +open. R3b must enforce one current-evidence gate for +replay/qualification/promotion/rollback/catalog fallback. Audit/read compatibility for unsafe legacy v1 proposals and public realistic fixtures remain R3c; the public paired-trial controller remains R5. Do not count current legacy positive fixtures as proof of that integration. No live providers, @@ -393,11 +427,11 @@ provider paths must not be advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Implement the R3b saved-run v2 outcome reader and current-evidence - eligibility. Join `experiment_specs` / `experiment_assignments` to actual - role attempts and final/correction outcome rows; recompute original input - and per-arm execution fingerprints from the immutable preparation witness. - Validate every relevant attempt, including fallbacks/repairs. Finish +2. Finish R3b.1 from the preserved partial reader: reproduce/repair honest + terminal failed-output handling and run the full unchanged-source gate. + Then implement R3b.2 shared current-evidence eligibility and append-only + evaluation/review revisions. Validate every relevant attempt, including + fallbacks/repairs. Finish R3b/R3c saved evidence, replay, qualification/promotion/rollback/catalog-fallback eligibility and historical compatibility, then execute R4–R6 in dependency order. Preserve R1/R2, diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 8a2f097..c7740bb 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -15,6 +15,15 @@ "status": "in_progress", "next_milestone": "M5", "next_work_package": "R3b.1", + "partial_implementation_checkpoint": { + "recorded_on": "2026-10-01", + "baseline_revision": "a22b949", + "status": "partial", + "artifact": "evidence/R3b1-reader-partial-2026-10-01.json", + "scope": "Authoritative v2 saved-run reader, replay invalidation, prelaunch controlled-input guard and public offline fixture; no install or provider calls", + "next_action": "Reproduce and repair honest terminal-failure stdout decoding, run full core integration, then R3b.2 shared current-evidence eligibility and append-only revisions", + "limitations": "Full integration and independent reader audit not complete; production paired-trial controller remains R5; native Gemini worker and refreshed installed proof remain unverified" + }, "planning_checkpoint": { "recorded_on": "2026-10-01", "inspected_checkpoint": "b766f9e", @@ -35,7 +44,7 @@ "review_work_packages": [ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, - {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3a-provenance-contract-2026-10-01.json", "checkpoint": "R3a source/offline verified at 672e383: outcome uniqueness, v2 profile/execution/input/assignment contracts, schema-14 fenced prelaunch persistence and normalized-chain evaluator; 47 focused, nine migration checks and complete 402-test gate pass with two optional-SDK skips. Review findings repaired and independently rechecked; SQLite warning remains R5. Public v2 saved-run reader and shared eligibility remain R3b; legacy/public/correction proof remains R3c. R3 is not closed."}, + {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3b1-reader-partial-2026-10-01.json", "checkpoint": "R3a remains verified at 672e383 with its complete 402-test gate. R3b.1 now has a partial public v2 saved-run reader, changed-evidence replay rejection and controlled-input launch guard. Imported-profile mismatch was reproduced and repaired. Honest terminal-failure decoding is an open inspection concern; full integration and independent reader audit are not complete. Shared eligibility and append-only revisions remain R3b.2; legacy/public compatibility remains R3c. R3 is not closed."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, {"id": "R6", "title": "Normal terminal experience and readiness", "status": "pending", "milestones": ["M5", "M7"], "depends_on": ["R1", "R2", "R4", "R5"], "items": ["G3", "G4"]}, diff --git a/docs/plans/engineering-team/evidence/R3b1-reader-partial-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b1-reader-partial-2026-10-01.json new file mode 100644 index 0000000..c4b85dc --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3b1-reader-partial-2026-10-01.json @@ -0,0 +1,61 @@ +{ + "schema_version": 1, + "work_package": "R3b.1", + "status": "partial", + "recorded_at": "2026-10-01T19:58:04Z", + "baseline_revision": "a22b949", + "implementation": [ + "Read v2 evidence from immutable preparation assignments and actual saved runs inside one ledger transaction", + "Validate preparation/event/attempt/profile/package/outcome identities and captured artifact bytes", + "Reject changed-evidence v2 replay while preserving historical evaluation bytes", + "Check controlled inputs and selected execution before reserving an assigned trial attempt" + ], + "source_sha256": { + "plugin/core/src/devsquad/store.py": "b474a0771150f4a38ba953a2fe13869fdaa7f60fb85ec15b0420e7a6d20ed847", + "plugin/core/src/devsquad/experiment_evidence.py": "d8c756fd18bdfdd77f3096385b09c872f7cf12a9d4dd66dda393f2eff3679d65", + "test/core/experiment_runtime_fixture.py": "eb653e4f60466bee52fbf33ffeb95353a79cdddc0f3c637f5b16a5186d0d32ed", + "test/core/test_experiment_saved_runs.py": "df6c0ac8f7e2b398727291945eab05a798ac1eceaacf4090890d62767fdbf507", + "test/core/test_experiment_evidence_integrity.py": "834d58e4827e96cafa1cc7053098fd11ed42fcb795fcad001f5b64cedbf77483" + }, + "verification": { + "public_reader_red": "One public evaluator test errored before integration after four real offline arms completed; missing saved-project identity; 5.527 seconds", + "imported_profile_red": "One deterministic test failed before semantic artifact binding: ContractError not raised; 1.505 seconds", + "earlier_focused_green": "59 tests passed in 43.570 seconds; extra prelaunch guard case then passed separately", + "checkpoint_focused_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_experiment_evidence_integrity test_experiment_saved_runs test_experiment_reuse test_experiment_provenance test_experiment_assignment_store test_experiment_v2_evaluation test_learning test_lifecycle -v", + "checkpoint_focused_result": "60 tests passed in 46.206 seconds, exit 0; process access enabled for cancellation and cleanup", + "bash_command": "bash test/run.sh", + "bash_result": "11 files, 227 assertions passed, exit 0; process access enabled", + "generated_reference": "python3 scripts/generate-core-reference.py --check passed", + "full_core_gate": "Not run for this partial reader source", + "independent_reader_review": "Incomplete; a bounded review raised an execution-binding concern but its follow-up hit a usage limit. No passed independent audit is claimed.", + "unavailable_earlier_handle": "A prior 50-test handle disappeared during a model switch; its result is unavailable and is not counted as a pass" + }, + "fixture_boundary": { + "positive_runs": "Public Service start, real offline worker/check subprocesses, host claim/complete and manual outcome addition", + "predeclaration": "A test-only preparation resolver seam attaches the frozen assignment before real complete_preparation; production paired-trial entry remains R5", + "metrics": "Synthetic host accept/reject dispositions; not native model-quality evidence", + "sql_tampering": "Negative corruption cases only; positive terminal runs are not fabricated with SQL" + }, + "open_findings": [ + { + "status": "inspection_concern_not_yet_reproduced", + "description": "Branch-review stdout validation currently treats absent output_metadata.failure as success. An honest terminal failed worker may lack that key and emit non-JSON output.", + "next_action": "Add a real offline terminal failed-reviewer regression; distinguish successful imported review evidence from opaque failed output, preserving failed exposure and diagnostics" + } + ], + "remaining": [ + "Resolve the terminal-failure concern and run full core integration before accepting R3b.1", + "R3b.2 shared eligibility for replay, qualification, promotion, rollback and catalog fallback", + "Explicit append-only evaluation/review revisions after new corrections", + "R3c public historical/legacy compatibility and realistic failure/repair proof", + "R5 production paired-trial controller and automatic terminal-outcome projection", + "R8 updated install and provider/surface acceptance" + ], + "gemini_antigravity_status": { + "evidence": "M7-installed-runtime-2026-09-29.json", + "verified": "Real read-only squad_status through Antigravity observed the terminal-created saved run", + "not_verified": "Gemini implementation/review worker execution and operation against the newly repaired installed runtime", + "user_action": "No additional Gemini login action currently recorded; preserve normal signed-in session" + }, + "external_actions": "No provider generation, API request, installation refresh, global settings change, purchase, reset redemption or push during this reader slice" +} diff --git a/plugin/core/src/devsquad/experiment_evidence.py b/plugin/core/src/devsquad/experiment_evidence.py new file mode 100644 index 0000000..e61e88c --- /dev/null +++ b/plugin/core/src/devsquad/experiment_evidence.py @@ -0,0 +1,344 @@ +"""Authoritative v2 experiment inputs from one consistent saved-ledger read. + +The pure evaluator accepts normalized witnesses, but callers cannot supply +them here. Assignment, execution and outcome identity come from the ledger. +The caller owns the transaction, including any subsequent decision mutation. +""" + +from __future__ import annotations + +from datetime import datetime +import hashlib +from pathlib import Path +import sqlite3 +from typing import Any + +from .contracts import ContractError +from .claude_identity import strict_json +from .experiment_provenance import ( + assignment_for, paired_input_identity, selected_execution_fingerprint, + validate_arm_chain, validate_assignment, +) +from .learning import validate_outcome +from .store import canonical_json + + +def _digest(value: Any) -> str: + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def _object(payload: str, label: str) -> dict[str, Any]: + try: + value = strict_json(payload) + except (ContractError, TypeError, ValueError) as exc: + raise ContractError(f"experiment saved {label} is not valid JSON") from exc + if not isinstance(value, dict): + raise ContractError(f"experiment saved {label} must be an object") + return value + + +def _outcome(row: sqlite3.Row, run: sqlite3.Row, now: datetime) -> dict[str, Any]: + value = validate_outcome(_object(row["payload_json"], "outcome"), now=now) + if (row["run_id"] != run["id"] or row["payload_sha256"] != _digest(value) + or any(row[key] != value[key] for key in ( + "outcome_id", "kind", "verdict", "selection_mode", + "observed_at", "corrects_outcome_id", + )) or value["selection_mode"] != "experimental"): + raise ContractError("experiment saved outcome identity/hash is inconsistent") + if value["kind"] == "final" and value["verdict"] != run["state"]: + raise ContractError("experiment final outcome does not match terminal run") + return value + + +def _artifact( + connection: sqlite3.Connection, artifact_id: str, run_id: str, +) -> tuple[dict[str, Any], bytes]: + row = connection.execute( + "SELECT id,run_id,name,path,sha256,byte_size FROM artifacts WHERE id=?", + (artifact_id,), + ).fetchone() + if row is None or row["run_id"] != run_id: + raise ContractError("experiment attempt artifact belongs to another run or is missing") + try: + content = Path(row["path"]).read_bytes() + except OSError as exc: + raise ContractError("experiment attempt artifact is unavailable") from exc + if len(content) != row["byte_size"] or hashlib.sha256(content).hexdigest() != row["sha256"]: + raise ContractError("experiment attempt artifact hash is inconsistent") + return {key: row[key] for key in ("id", "name", "sha256", "byte_size")}, content + + +def verify_prelaunch_snapshot( + connection: sqlite3.Connection, *, run_id: str, + snapshot: Any, package_digest: str, +) -> None: + """Fence continued trial execution to its original controlled inputs.""" + frozen = connection.execute( + "SELECT a.*,r.package_digest AS run_package,p.git_common_dir " + "FROM experiment_assignments a JOIN runs r ON r.id=a.run_id " + "JOIN projects p ON p.id=r.project_id WHERE a.run_id=?", (run_id,), + ).fetchone() + if frozen is None: + if isinstance(snapshot, dict) and ( + snapshot.get("experiment_spec") is not None + or snapshot.get("experiment_assignment") is not None): + raise ContractError("experiment launch has no immutable prelaunch assignment") + return + if not isinstance(snapshot, dict): + raise ContractError("experiment launch snapshot is missing") + original = _object(frozen["snapshot_json"], "preparation snapshot") + spec = original.get("experiment_spec") + assignment = validate_assignment( + _object(frozen["assignment_json"], "assignment"), spec=spec, + project_common_dir=frozen["git_common_dir"], + ) + if (package_digest != frozen["package_digest"] + or package_digest != frozen["run_package"] + or _digest(assignment) != frozen["assignment_sha256"] + or original.get("experiment_assignment") != assignment + or snapshot.get("experiment_spec") != spec + or snapshot.get("experiment_assignment") != assignment): + raise ContractError("experiment launch differs from its frozen assignment/package") + role = assignment["role"] + inputs = paired_input_identity(snapshot, role=role, package_digest=package_digest) + selected = snapshot["routing"]["roles"][role]["selected"] + if (any(inputs[key] != assignment[key] for key in inputs) + or any(selected[key] != assignment[key] for key in ("profile_id", "profile_sha256")) + or selected_execution_fingerprint(snapshot, role=role) != assignment["execution_sha256"]): + raise ContractError("experiment launch changed its controlled input/profile/execution") + + +def _attempts( + connection: sqlite3.Connection, run: sqlite3.Row, frozen: sqlite3.Row, + snapshot: dict[str, Any], assignment: dict[str, Any], +) -> tuple[list[dict[str, Any]], bool]: + """Validate *all* reservations/history, not only a final successful arm. + + Unstarted reservations remain in the digest, but do not become exposure. + A launched attempt without reconciled output keeps the arm unavailable. + """ + events = connection.execute( + "SELECT run_version,type,payload FROM events WHERE run_id=? ORDER BY run_version", + (run["id"],), + ).fetchall() + prepared_version = frozen["frozen_run_version"] + if (type(prepared_version) is not int or prepared_version < 2 + or type(frozen["preparation_fencing_token"]) is not int + or frozen["preparation_fencing_token"] < 1 + or run["version"] < prepared_version): + raise ContractError("experiment preparation fence/version is invalid") + queued = [event for event in events if event["run_version"] == prepared_version] + if len(queued) != 1 or queued[0]["type"] != "run.queued": + raise ContractError("experiment preparation has no saved queued fence") + if not events or events[0]["type"] != "run.preparing": + raise ContractError("experiment preparation history is unavailable") + preparation_token = 1 + claimed, launched, unstarted = {}, {}, set() + for event in events: + if event["type"] == "run.preparation_reclaimed" and event["run_version"] < prepared_version: + preparation_token = _object(event["payload"], "preparation event").get("fencing_token") + if event["type"] not in {"supervisor.claimed", "run.running", "run.unstarted_attempt_recovered"}: + continue + payload = _object(event["payload"], "attempt event") + attempt_id = payload.get("attempt_id") + if not isinstance(attempt_id, str) or not attempt_id: + raise ContractError("experiment attempt event identity is missing") + if event["type"] == "run.unstarted_attempt_recovered": + unstarted.add(attempt_id) + continue + target = claimed if event["type"] == "supervisor.claimed" else launched + if attempt_id in target or event["run_version"] <= prepared_version: + raise ContractError("experiment assignment must precede distinct attempt events") + target[attempt_id] = {"run_version": event["run_version"], **payload} + if preparation_token != frozen["preparation_fencing_token"]: + raise ContractError("experiment preparation fencing token does not match history") + rows = connection.execute("SELECT * FROM attempts WHERE run_id=?", (run["id"],)).fetchall() + if set(claimed) != {row["id"] for row in rows} or not set(launched) <= set(claimed): + raise ContractError("experiment attempt reservations do not match saved history") + history = [] + exposed = False + incomplete = False + for row in sorted(rows, key=lambda item: claimed[item["id"]]["run_version"]): + role = row["role"] + route = snapshot["routing"]["roles"].get(role) + if (row["project_id"] != run["project_id"] + or row["package_digest"] != frozen["package_digest"] + or not isinstance(route, dict)): + raise ContractError("experiment actual attempt project/package/role is inconsistent") + slots = [route["selected"], *route.get("fallbacks", [])] + index = row["profile_index"] + if (type(index) is not int or not 0 <= index < len(slots) + or row["profile_id"] != slots[index]["profile_id"] + or row["account_pool_id"] != slots[index]["profile"]["account_pool_id"]): + raise ContractError("experiment actual attempt profile does not match its frozen slot") + selected = slots[index] + started = launched.get(row["id"]) + if (row["pid"] is None) != (started is None): + raise ContractError("experiment actual attempt launch identity is inconsistent") + if started is not None: + if (started["run_version"] <= claimed[row["id"]]["run_version"] + or any(started.get(key) != row[key] for key in ("pid", "pgid", "process_start_id"))): + raise ContractError("experiment actual attempt launch fence is inconsistent") + if row["status"] not in {"finished", "recovery_required"}: + incomplete = True + # A repaired pre-gate crash never became a worker exposure. All other + # launched tested-role attempts, including failed fallbacks, must be + # the explicitly declared execution, not merely an eligible profile. + was_worker = started is not None and row["id"] not in unstarted + if was_worker and role == assignment["role"]: + actual = {**snapshot, "routing": {**snapshot["routing"], "roles": { + **snapshot["routing"]["roles"], role: {**route, "selected": selected}, + }}} + if (selected["profile_id"] != assignment["profile_id"] + or selected["profile_sha256"] != assignment["profile_sha256"] + or selected_execution_fingerprint(actual, role=role) != assignment["execution_sha256"]): + raise ContractError("experiment actual attempt does not execute the declared arm") + metadata = None + artifacts = [] + captures = {} + if row["output_metadata"] is not None: + metadata = _object(row["output_metadata"], "attempt output") + if started is None or row["id"] in unstarted: + raise ContractError("experiment unstarted attempt cannot have worker output") + for stream in ("stdout", "stderr"): + artifact, content = _artifact(connection, row[f"{stream}_artifact_id"], run["id"]) + capture = metadata.get(stream) + if (not isinstance(capture, dict) + or capture.get("captured_sha256") != artifact["sha256"] + or capture.get("captured_bytes") != artifact["byte_size"]): + raise ContractError("experiment attempt output does not match its saved capture") + artifacts.append(artifact) + captures[stream] = content + if was_worker and role == assignment["role"] and row["status"] == "finished": + exposed = True + elif was_worker: + incomplete = True + # These immutable artifacts retain imported observed identity and + # native usage; failure diagnostics are retained in output_metadata. + for prefix in ("review", "implementation", "lead"): + artifact_row = connection.execute( + "SELECT id FROM artifacts WHERE run_id=? AND name=?", + (run["id"], f"{prefix}-attempt-{row['id']}.json"), + ).fetchone() + if artifact_row is not None: + artifact, content = _artifact(connection, artifact_row["id"], run["id"]) + document = _object(content, "imported attempt evidence") + evidence = document.get("attempt", document) + if (not isinstance(evidence, dict) or evidence.get("role") != role + or canonical_json(evidence.get("selected_profile")) != canonical_json(selected)): + raise ContractError("experiment imported execution differs from its frozen attempt profile") + artifacts.append(artifact) + if (role == "reviewer" and snapshot["task"]["workflow"] == "branch-review" + and captures and metadata.get("failure") is None): + from .workflows import validate_branch_review_evidence + + validate_branch_review_evidence(strict_json(captures["stdout"]), snapshot) + history.append({ + **{key: row[key] for key in ( + "id", "run_id", "project_id", "role", "status", "profile_id", + "profile_index", "account_pool_id", "package_digest", "pid", + "pgid", "process_start_id", "created_at", "finished_at", + )}, + "reservation": claimed[row["id"]], "launch": started, + "unstarted_recovered": row["id"] in unstarted, + "profile_sha256": selected["profile_sha256"], + "output_metadata": metadata, "artifacts": artifacts, + }) + return history, exposed and not incomplete + + +def read_experiment_chains( + connection: sqlite3.Connection, *, spec: dict[str, Any], project_id: str | None, + project_common_dir: str, evaluated_at: str, +) -> dict[str, dict[str, Any]]: + """Build v2 witnesses solely from immutable preparation and durable runs.""" + if not connection.in_transaction: + raise ContractError("experiment evidence requires a consistent ledger transaction") + if spec["schema_version"] != 2: + raise ContractError("saved-run provenance requires a v2 experiment") + declared = connection.execute( + "SELECT project_id,spec_json,spec_sha256 FROM experiment_specs WHERE experiment_id=?", + (spec["experiment_id"],), + ).fetchone() + if (declared is None or declared["project_id"] != project_id + or declared["spec_json"] != canonical_json(spec) + or declared["spec_sha256"] != _digest(spec)): + raise ContractError("experiment specification is not predeclared for this saved project") + records = connection.execute( + "SELECT * FROM experiment_assignments WHERE experiment_id=?", (spec["experiment_id"],), + ).fetchall() + expected_keys = {(case["case_id"], arm) for case in spec["cases"] for arm in ("control", "candidate")} + frozen_by_arm = {(row["case_id"], row["arm"]): row for row in records} + if len(frozen_by_arm) != len(records) or not set(frozen_by_arm) <= expected_keys: + raise ContractError("experiment saved assignments have undeclared or duplicate arms") + chains = {} + for case in spec["cases"]: + for arm in ("control", "candidate"): + expected = assignment_for(spec, case["case_id"], arm, project_common_dir=project_common_dir) + final_row = connection.execute( + "SELECT * FROM outcomes WHERE outcome_id=?", (expected["outcome_id"],), + ).fetchone() + frozen = frozen_by_arm.get((case["case_id"], arm)) + if frozen is None: + if final_row is not None: + raise ContractError("experiment outcome has no prelaunch assignment") + continue + assignment = validate_assignment( + _object(frozen["assignment_json"], "assignment"), spec=spec, + project_common_dir=project_common_dir, + ) + if (assignment != expected or frozen["assignment_sha256"] != _digest(assignment) + or frozen["outcome_id"] != expected["outcome_id"]): + raise ContractError("experiment saved assignment identity/hash is inconsistent") + run = connection.execute( + "SELECT r.*,p.git_common_dir FROM runs r JOIN projects p ON p.id=r.project_id WHERE r.id=?", + (frozen["run_id"],), + ).fetchone() + if (run is None or run["project_id"] != project_id + or run["git_common_dir"] != project_common_dir + or run["package_digest"] != frozen["package_digest"]): + raise ContractError("experiment assigned run project/package is inconsistent") + snapshot = _object(frozen["snapshot_json"], "preparation snapshot") + if (snapshot.get("experiment_spec") != spec + or snapshot.get("experiment_assignment") != assignment): + raise ContractError("experiment immutable preparation witness is inconsistent") + inputs = paired_input_identity(snapshot, role=assignment["role"], package_digest=frozen["package_digest"]) + selected = snapshot["routing"]["roles"][assignment["role"]]["selected"] + execution = selected_execution_fingerprint(snapshot, role=assignment["role"]) + if (any(inputs[key] != assignment[key] for key in inputs) + or any(selected[key] != assignment[key] for key in ("profile_id", "profile_sha256")) + or execution != assignment["execution_sha256"]): + raise ContractError("experiment frozen profile/execution/input differs from declaration") + history, exposed = _attempts(connection, run, frozen, snapshot, assignment) + outcomes = connection.execute( + "SELECT * FROM outcomes WHERE run_id=? ORDER BY observed_at,id", (run["id"],), + ).fetchall() + if any(row["kind"] == "final" and row["outcome_id"] != expected["outcome_id"] for row in outcomes): + raise ContractError("experiment run final outcome differs from its assigned outcome") + if final_row is None: + continue + if run["state"] not in {"succeeded", "failed", "cancelled"} or final_row["kind"] != "final": + raise ContractError("experiment outcome requires a terminal assigned run") + final = _outcome(final_row, run, datetime.fromisoformat(evaluated_at)) + corrections = [ + _outcome(row, run, datetime.fromisoformat(evaluated_at)) + for row in outcomes if row["kind"] == "late_correction" + ] + if not exposed: + continue + chain = { + "final": final, "late_corrections": corrections, + "provenance": { + "run_id": run["id"], "assignment": assignment, + "profile_sha256": selected["profile_sha256"], + "execution_sha256": execution, **inputs, + "attempt_ids": [attempt["id"] for attempt in history], + "attempts_sha256": _digest(history), + "final_outcome_sha256": _digest(final), + "correction_sha256": [_digest(value) for value in corrections], + }, + } + validate_arm_chain(chain, spec=spec, case=case, arm=arm, + project_common_dir=project_common_dir, evaluated_at=evaluated_at) + chains[expected["outcome_id"]] = chain + return chains diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index bc1baf3..27e5480 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -1768,7 +1768,10 @@ def evaluate_learning_experiment( spec = validate_experiment(experiment) project_path = Path(spec["project_path"]).resolve(strict=True) - spec = {**spec, "project_path": str(project_path)} + # V2 hashes the exact predeclared specification. Resolving a symlink + # for project lookup must not rewrite that declaration after launch. + if spec["schema_version"] == 1: + spec = {**spec, "project_path": str(project_path)} spec_json = canonical_json(spec) spec_sha256 = hashlib.sha256(spec_json.encode()).hexdigest() current = _authoritative_now(now) @@ -1780,11 +1783,11 @@ def evaluate_learning_experiment( "FROM experiments WHERE experiment_id=?", (spec["experiment_id"],), ).fetchone() - if existing is not None: - if existing["spec_json"] != spec_json: - raise ConflictError( - "experiment id was already used with a different specification", - ) + if existing is not None and existing["spec_json"] != spec_json: + raise ConflictError( + "experiment id was already used with a different specification", + ) + if existing is not None and spec["schema_version"] == 1: self.connection.execute("COMMIT") return { "experiment": spec, @@ -1797,33 +1800,64 @@ def evaluate_learning_experiment( "SELECT id FROM projects WHERE git_common_dir=?", (str(common_dir),), ).fetchone() project_id = project["id"] if project is not None else None - records = [] - if project_id is not None: - records = self.connection.execute( - "SELECT o.payload_json FROM outcomes o " - "JOIN runs r ON r.id=o.run_id WHERE r.project_id=? " - "ORDER BY o.observed_at,o.id", - (project_id,), - ).fetchall() chains: dict[str, dict[str, Any]] = {} - corrections: dict[str, list[dict[str, Any]]] = {} - for record in records: - outcome = json.loads(record["payload_json"]) - if outcome["kind"] == "final": - chains[outcome["outcome_id"]] = { - "final": outcome, - "late_corrections": [], - } - else: - corrections.setdefault( - outcome["corrects_outcome_id"], [], - ).append(outcome) - for outcome_id, history in corrections.items(): - if outcome_id in chains: - chains[outcome_id]["late_corrections"] = history + if spec["schema_version"] == 2: + from .experiment_evidence import read_experiment_chains + + chains = read_experiment_chains( + self.connection, spec=spec, project_id=project_id, + project_common_dir=str(common_dir), evaluated_at=current.isoformat(), + ) + else: + # Historical v1 decoding remains separate from v2 authority. + records = [] + if project_id is not None: + records = self.connection.execute( + "SELECT o.payload_json FROM outcomes o " + "JOIN runs r ON r.id=o.run_id WHERE r.project_id=? " + "ORDER BY o.observed_at,o.id", + (project_id,), + ).fetchall() + corrections: dict[str, list[dict[str, Any]]] = {} + for record in records: + outcome = json.loads(record["payload_json"]) + if outcome["kind"] == "final": + chains[outcome["outcome_id"]] = { + "final": outcome, "late_corrections": [], + } + else: + corrections.setdefault( + outcome["corrects_outcome_id"], [], + ).append(outcome) + for outcome_id, history in corrections.items(): + if outcome_id in chains: + chains[outcome_id]["late_corrections"] = history evaluation = evaluate_experiment( spec, chains, evaluated_at=current.isoformat(), + project_common_dir=str(common_dir), ) + if existing is not None: + try: + saved = json.loads(existing["evaluation_json"]) + except (ValueError, TypeError) as exc: + raise ContractError("saved experiment evaluation is invalid") from exc + if (not isinstance(saved, dict) + or saved.get("evaluated_at") != existing["recorded_at"] + or hashlib.sha256(existing["evaluation_json"].encode()).hexdigest() + != existing["evaluation_sha256"]): + raise ContractError("saved experiment evaluation hash/identity is invalid") + # Receipts are immutable. Corrections or changed run evidence + # require an explicit revision, never a silently updated replay. + if canonical_json({**evaluation, "evaluated_at": saved["evaluated_at"]}) != canonical_json(saved): + raise ContractError( + "experiment evidence changed; saved evaluation is stale and requires explicit review/revision", + ) + self.connection.execute("COMMIT") + return { + "experiment": spec, "evaluation": saved, + "evaluation_sha256": existing["evaluation_sha256"], + "recorded_at": existing["recorded_at"], "replayed": True, + } evaluation_json = canonical_json(evaluation) evaluation_sha256 = hashlib.sha256(evaluation_json.encode()).hexdigest() recorded_at = current.isoformat() @@ -2776,6 +2810,16 @@ def reserve_attempt( raise ConflictError("run already has a supervisor claim") if not run["worktree_path"]: raise ContractError("run has no canonical worktree identity") + from .experiment_evidence import verify_prelaunch_snapshot + + try: + current_snapshot = json.loads(run["mutable_snapshot"] or "{}") + except (TypeError, ValueError) as exc: + raise ContractError("frozen run snapshot is invalid") from exc + verify_prelaunch_snapshot( + self.connection, run_id=run_id, snapshot=current_snapshot, + package_digest=package_digest, + ) self._enforce_attempt_budget(run_id, run) if account_pool_id is not None: if not isinstance(account_pool_id, str) or not account_pool_id: diff --git a/test/core/experiment_runtime_fixture.py b/test/core/experiment_runtime_fixture.py new file mode 100644 index 0000000..2cd097c --- /dev/null +++ b/test/core/experiment_runtime_fixture.py @@ -0,0 +1,209 @@ +"""Real offline experiment runs with a test-only preflight assignment seam. + +This is not a public experiment controller or native-provider quality proof. +Profiles and reviews are explicit fixtures, but preparation, worker processes, +checks, host decisions and outcomes use the actual Service/Store workflow. +""" + +from __future__ import annotations + +from contextlib import ExitStack +import copy +from datetime import datetime, timezone +from pathlib import Path +import subprocess +import sys +import time +from unittest.mock import patch + +from test_experiment_provenance import digest, execution_digest +from test_learning import experimental_final +from test_lifecycle import profile, review_task, routing_policy + +from devsquad.experiment_provenance import assignment_for, paired_input_identity +from devsquad.router import load_routing +from devsquad.service import Service +from devsquad.store import Store, canonical_json, git_common_dir, request_hash + + +class ExperimentRuntimeFixture: + def __init__(self, root: Path, *, with_fallback=False): + self.root = root.resolve() + self.root.mkdir(parents=True, exist_ok=True) + self.with_fallback = with_fallback + self.repo = self.root / "repo" + self.service = Service(self.root / "runtime") + self.runs = {} + self.git("init", "-q", str(self.repo), outside=True) + self.git("config", "user.email", "test@example.invalid") + self.git("config", "user.name", "Test") + (self.repo / "README").write_text("base\n") + self.git("add", "README") + self.git("commit", "-qm", "baseline") + self.base = self.git("rev-parse", "HEAD").strip() + self.targets = {} + for case_id in ("eval-1", "hold-1"): + (self.repo / "README").write_text(f"candidate {case_id}\n") + self.git("add", "README") + self.git("commit", "-qm", f"candidate {case_id}") + self.targets[case_id] = self.git("rev-parse", "HEAD").strip() + self.common = str(git_common_dir(self.repo)) + self.profiles = { + arm: profile(f"profile-{letter}", f"model-{letter}") + for arm, letter in (("control", "a"), ("candidate", "b")) + } + if with_fallback: + # The real offline worker deliberately errors on this suffix. + self.profiles["candidate"]["id"] = "candidate-fixture-fail" + self.registry = { + "schema_version": 1, "profiles": list(self.profiles.values()), + "bindings": {"review.deep": {"profile_id": "profile-a", "version": 7}}, + } + self.policy = routing_policy() + if with_fallback: + fallback = profile("profile-fallback", "model-fallback") + self.registry["profiles"].append(fallback) + self.policy["roles"]["reviewer"] = [{"kind": "profile", "id": fallback["id"]}] + self.package_path, self.package_digest = self.service._freeze_package() + cases = [] + for case_id, split in (("eval-1", "evaluation"), ("hold-1", "held_out")): + identities = paired_input_identity( + self.declaration_snapshot(case_id, "control"), + role="reviewer", package_digest=self.package_digest, + ) + cases.append({ + "case_id": case_id, "split": split, + "control_outcome_id": f"control-{case_id}", + "candidate_outcome_id": f"candidate-{case_id}", + **identities, + }) + self.spec = { + "schema_version": 2, "experiment_id": "saved-run-review-pair", + "project_path": str(self.repo), + "question": "Does the candidate fixture improve paired acceptance?", + "hypothesis": "Candidate outcomes improve in evaluation and held-out cases.", + "evidence_availability": "tracked_fixture", + "variable": { + "kind": "profile_binding", "alias": "review.deep", "role": "reviewer", + **{f"{arm}_profile_id": value["id"] for arm, value in self.profiles.items()}, + **{f"{arm}_profile_sha256": digest(value) for arm, value in self.profiles.items()}, + **{f"{arm}_execution_sha256": execution_digest(value) for arm, value in self.profiles.items()}, + }, + "cases": cases, + "gate": { + "min_evaluation_pairs": 1, "min_held_out_pairs": 1, + "noninferiority_margin": 0.0, "minimum_success_gain": 1.0, + "max_candidate_escaped_defects": 0, + }, + "budget": {"max_cases": 2, "max_worker_invocations": 8 if with_fallback else 4, "wall_seconds": 600}, + "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, + } + + def git(self, *args, outside=False): + command = ["git"] if outside else ["git", "-C", str(self.repo)] + return subprocess.run(command + list(args), check=True, capture_output=True, text=True).stdout + + def task(self, case_id, arm): + task = review_task(self.repo) + task["project"].update({"base_ref": self.base, "target_ref": self.targets[case_id]}) + task["goal"] = "Review the exact README candidate and pass the declared check." + task["routing"] = { + "profiles": copy.deepcopy(self.registry), "policy": copy.deepcopy(self.policy), + "overrides": {"reviewer": { + "profile_id": self.profiles[arm]["id"], + "fallback": "policy" if self.with_fallback else "none", + }}, + } + task["checks"] = [{ + "id": "read-candidate", "argv": [sys.executable, "-c", "from pathlib import Path; assert Path('README').read_text().startswith('candidate')"], + "cwd": ".", "timeout_seconds": 10, "required_to_pass": True, + }] + task["budget"]["wall_seconds"] = 120 + if self.with_fallback: + task["budget"].update(max_worker_invocations=2, max_fallbacks_per_step=1) + return task + + def declaration_snapshot(self, case_id, arm): + task = self.task(case_id, arm) + identity = { + "schema_version": 1, "base_oid": self.base, + "target_oid": self.targets[case_id], "changed_paths": ["README"], + } + return { + "task": task, "base_oid": self.base, "target_oid": self.targets[case_id], + "workspace": {**identity, "candidate_sha256": digest(identity)}, + "configs": {"policy_file": {"sha256": digest(self.policy)}}, + "routing": load_routing(task, canonical_json(self.registry), canonical_json(self.policy)), + } + + def store(self): + return Store(self.service.database, self.service.artifacts) + + def wait(self, run_id): + deadline = time.monotonic() + 20 + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] in {"awaiting_host", "failed", "succeeded", "cancelled"}: + return status + time.sleep(0.05) + raise AssertionError(f"saved-run fixture did not reach a gate: {self.service.status(run_id)}") + + def run_arm(self, case_id, arm, *, no_attempt=False): + original = self.service._resolve_snapshot + + def predeclared_assignment(*args, **kwargs): + snapshot = original(*args, **kwargs) + snapshot["experiment_spec"] = copy.deepcopy(self.spec) + snapshot["experiment_assignment"] = assignment_for( + self.spec, case_id, arm, project_common_dir=self.common, + ) + return snapshot + + with ExitStack() as stack: + stack.enter_context(patch.object(self.service, "_resolve_snapshot", side_effect=predeclared_assignment)) + if no_attempt: + stack.enter_context(patch.object(self.service, "_spawn_daemon", return_value=0)) + started = self.service.start( + self.task(case_id, arm), f"{arm}-{case_id}", + _internal_review_fixture={"verdict": "clean", "summary": "Fixture review of the frozen candidate.", "findings": []}, + ) + run_id = started["run_id"] + self.runs[(case_id, arm)] = run_id + if started["state"] != "queued": + raise AssertionError(f"public fixture preparation failed: {started}") + if no_attempt: + self.service.cancel(run_id) + verdict = "cancelled" + else: + waiting = self.wait(run_id) + if waiting["state"] != "awaiting_host": + raise AssertionError(f"public fixture worker failed: {waiting}") + claimed = self.service.handoff_claim(run_id, waiting["version"], "experiment-fixture-host") + packet = claimed["handoff"]["packet"] + body = { + "schema_version": 1, "submission_id": f"finish-{arm}-{case_id}", + "disposition": "accept" if arm == "candidate" else "reject", + "reason": "Predeclared offline fixture disposition.", + "evidence_refs": [{"artifact_id": reference["artifact_id"], "sha256": reference["sha256"]} for reference in packet["artifacts"]], + } + completed = self.service.handoff_complete(run_id, claimed["claim"], {**body, "submission_hash": request_hash(body)}) + verdict = "succeeded" if arm == "candidate" else "failed" + if completed["state"] != verdict: + raise AssertionError(f"public fixture disposition failed: {completed}") + outcome = experimental_final(f"{arm}-{case_id}", verdict) + outcome["observed_at"] = datetime.now(timezone.utc).isoformat() + outcome["evidence_refs"] = ["receipt.json", "result-receipt.json"] + self.service.outcome_add(run_id, outcome) + return run_id + + def run_all(self, *, skip=None, no_attempt=None): + for case in self.spec["cases"]: + for arm in ("control", "candidate"): + key = (case["case_id"], arm) + if key != skip: + self.run_arm(*key, no_attempt=key == no_attempt) + + def close(self): + for run_id in self.runs.values(): + if self.service.status(run_id)["state"] not in {"succeeded", "failed", "cancelled"}: + self.service.cancel(run_id) diff --git a/test/core/test_experiment_evidence_integrity.py b/test/core/test_experiment_evidence_integrity.py new file mode 100644 index 0000000..4ca4558 --- /dev/null +++ b/test/core/test_experiment_evidence_integrity.py @@ -0,0 +1,180 @@ +"""Corrupt only negative copies of otherwise publicly executed arm evidence.""" + +import copy +import hashlib +import json +from pathlib import Path +import tempfile +import unittest +from unittest.mock import patch + +from experiment_runtime_fixture import ExperimentRuntimeFixture +from devsquad.contracts import ContractError +from devsquad.experiment_evidence import read_experiment_chains +from devsquad.experiment_provenance import assignment_for +from devsquad.store import canonical_json +from test_experiment_provenance import digest + + +class ExperimentEvidenceIntegrityTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-evidence-integrity-") + self.addCleanup(self.temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(self.temporary.name)) + self.addCleanup(self.fixture.close) + self.run_id = self.fixture.run_arm("eval-1", "candidate") + self.store = self.fixture.store() + self.addCleanup(self.store.close) + + def assert_rejected_without_persistence(self): + with self.assertRaises(ContractError): + self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiments").fetchone()[0], 0) + + def test_fence_event_and_outcome_corruptions_cannot_be_hidden_by_missing_partner(self): + assignment = dict(self.store.connection.execute( + "SELECT * FROM experiment_assignments WHERE run_id=?", (self.run_id,), + ).fetchone()) + run = self.store.run(self.run_id) + outcome = dict(self.store.connection.execute( + "SELECT * FROM outcomes WHERE run_id=?", (self.run_id,), + ).fetchone()) + queued = dict(self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? AND run_version=?", + (self.run_id, assignment["frozen_run_version"]), + ).fetchone()) + claimed = dict(self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? AND type='supervisor.claimed'", (self.run_id,), + ).fetchone()) + running = dict(self.store.connection.execute( + "SELECT * FROM events WHERE run_id=? AND type='run.running'", (self.run_id,), + ).fetchone()) + changed_launch = json.loads(running["payload"]) + changed_launch["pid"] += 1 + mutations = [ + ("fence_version", "experiment_assignments", "run_id", self.run_id, + "frozen_run_version", 1, assignment["frozen_run_version"]), + ("fence_token", "experiment_assignments", "run_id", self.run_id, + "preparation_fencing_token", 999, assignment["preparation_fencing_token"]), + ("assignment_hash", "experiment_assignments", "run_id", self.run_id, + "assignment_sha256", "0" * 64, assignment["assignment_sha256"]), + ("queued_event", "events", "id", queued["id"], "type", "ignored", queued["type"]), + ("claim_event", "events", "id", claimed["id"], "type", "ignored", claimed["type"]), + ("launch_event", "events", "id", running["id"], "payload", canonical_json(changed_launch), running["payload"]), + ("final_hash", "outcomes", "id", outcome["id"], "payload_sha256", "0" * 64, outcome["payload_sha256"]), + ("final_row_verdict", "outcomes", "id", outcome["id"], "verdict", "failed", outcome["verdict"]), + ("run_package", "runs", "id", self.run_id, "package_digest", "0" * 64, run["package_digest"]), + ("run_terminal", "runs", "id", self.run_id, "state", "failed", run["state"]), + ] + for name, table, key, identity, column, changed, original in mutations: + with self.subTest(mutation=name): + statement = f"UPDATE {table} SET {column}=? WHERE {key}=?" + self.store.connection.execute(statement, (changed, identity)) + try: + self.assert_rejected_without_persistence() + finally: + self.store.connection.execute(statement, (original, identity)) + self.assertEqual(self.fixture.service.policy_evaluate(self.fixture.spec)["evaluation"]["verdict"], "no_change") + + def test_saved_output_bytes_must_match_the_durable_capture(self): + artifact = self.store.connection.execute( + "SELECT f.path FROM attempts a JOIN artifacts f ON f.id=a.stdout_artifact_id WHERE a.run_id=?", + (self.run_id,), + ).fetchone() + path = Path(artifact["path"]) + original = path.read_bytes() + try: + path.write_bytes(original + b"corrupt") + self.assert_rejected_without_persistence() + finally: + path.write_bytes(original) + + def test_imported_profile_drift_is_rejected_even_with_a_valid_artifact_hash(self): + row = dict(self.store.connection.execute( + "SELECT * FROM artifacts WHERE run_id=? AND name LIKE 'review-attempt-%.json'", + (self.run_id,), + ).fetchone()) + path = Path(row["path"]) + original = path.read_bytes() + changed = json.loads(original) + selected = changed["selected_profile"] + selected["profile"]["model_id"] = "a-different-executed-model" + selected["profile_sha256"] = digest(selected["profile"]) + content = (canonical_json(changed) + "\n").encode() + # Simulate internally hash-consistent imported identity drift. A file + # checksum alone cannot bind its selected profile to the original arm. + try: + path.write_bytes(content) + self.store.connection.execute( + "UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (hashlib.sha256(content).hexdigest(), len(content), row["id"]), + ) + self.assert_rejected_without_persistence() + finally: + path.write_bytes(original) + self.store.connection.execute( + "UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (row["sha256"], row["byte_size"], row["id"]), + ) + + def test_trial_reservation_rejects_changed_inputs_before_any_worker(self): + resolve = self.fixture.service._resolve_snapshot + + def assigned(*args, **kwargs): + snapshot = resolve(*args, **kwargs) + snapshot["experiment_spec"] = copy.deepcopy(self.fixture.spec) + snapshot["experiment_assignment"] = assignment_for( + self.fixture.spec, "hold-1", "control", project_common_dir=self.fixture.common, + ) + return snapshot + + with (patch.object(self.fixture.service, "_resolve_snapshot", side_effect=assigned), + patch.object(self.fixture.service, "_spawn_daemon", return_value=0)): + started = self.fixture.service.start( + self.fixture.task("hold-1", "control"), "reserved-negative", + _internal_review_fixture={"verdict": "clean", "summary": "Fixture.", "findings": []}, + ) + run_id = started["run_id"] + self.fixture.runs[("hold-1", "control")] = run_id + self.assertEqual(started["state"], "queued") + run = self.store.run(run_id) + original = json.loads(run["mutable_snapshot"]) + selected = original["routing"]["roles"]["reviewer"]["selected"] + changed_profile = copy.deepcopy(original) + slot = changed_profile["routing"]["roles"]["reviewer"]["selected"] + slot["profile"]["model_id"] = "changed-after-preparation" + slot["profile_sha256"] = digest(slot["profile"]) + changed_task = copy.deepcopy(original) + changed_task["task"]["goal"] = "A different task after preparation." + for changed in (changed_profile, changed_task, []): + with self.subTest(snapshot_type=type(changed).__name__): + self.store.connection.execute( + "UPDATE runs SET mutable_snapshot=? WHERE id=?", (canonical_json(changed), run_id), + ) + try: + with self.assertRaisesRegex(ContractError, "experiment launch"): + self.store.reserve_attempt( + run_id, run["version"], "negative-worker", run["package_digest"], "reviewer", + account_pool_id=selected["profile"]["account_pool_id"], + profile_id=selected["profile_id"], profile_index=0, + ) + self.assertEqual(self.store.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()[0], 0) + finally: + self.store.connection.execute( + "UPDATE runs SET mutable_snapshot=? WHERE id=?", (run["mutable_snapshot"], run_id), + ) + + def test_reader_requires_one_consistent_transaction(self): + self.assertFalse(self.store.connection.in_transaction) + with self.assertRaisesRegex(ContractError, "consistent ledger transaction"): + read_experiment_chains( + self.store.connection, spec=self.fixture.spec, + project_id=self.store.run(self.run_id)["project_id"], + project_common_dir=self.fixture.common, evaluated_at="2026-10-01T00:00:00Z", + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_experiment_saved_runs.py b/test/core/test_experiment_saved_runs.py new file mode 100644 index 0000000..a561b4a --- /dev/null +++ b/test/core/test_experiment_saved_runs.py @@ -0,0 +1,230 @@ +"""Saved-run v2 evaluation through real offline workers and public completion.""" + +from datetime import datetime, timezone +import json +import tempfile +from pathlib import Path +import unittest + +from experiment_runtime_fixture import ExperimentRuntimeFixture +from test_experiment_provenance import digest +from test_learning import experimental_final + +from devsquad.contracts import ContractError +from devsquad.store import canonical_json + + +class ExperimentSavedRunsTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-saved-experiment-") + self.addCleanup(self.temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(self.temporary.name)) + self.addCleanup(self.fixture.close) + + def test_disjoint_public_runs_supply_v2_evaluation_without_submitted_provenance(self): + self.fixture.run_all() + self.assertEqual(len(set(self.fixture.runs.values())), 4) + store = self.fixture.store() + try: + rows = store.connection.execute("SELECT id,status,role,profile_id FROM attempts").fetchall() + self.assertEqual(len(rows), 4) + self.assertTrue(all(row["status"] == "finished" and row["role"] == "reviewer" for row in rows)) + self.assertEqual({row["profile_id"] for row in rows}, {"profile-a", "profile-b"}) + self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM experiment_assignments").fetchone()[0], 4) + finally: + store.close() + result = self.fixture.service.policy_evaluate(self.fixture.spec) + evaluation = result["evaluation"] + self.assertEqual(evaluation["schema_version"], 2) + self.assertEqual(evaluation["verdict"], "promotion_proposal") + self.assertEqual(evaluation["metrics"]["evaluation"]["available_pairs"], 1) + self.assertEqual(evaluation["metrics"]["held_out"]["available_pairs"], 1) + self.assertEqual(len(evaluation["evidence_sha256"]), 64) + replay = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertTrue(replay["replayed"]) + self.assertEqual(replay, {**result, "replayed": True}) + + def test_project_symlink_alias_preserves_the_predeclared_spec_hash(self): + alias = self.fixture.root / "project-alias" + alias.symlink_to(self.fixture.repo, target_is_directory=True) + self.fixture.spec["project_path"] = str(alias) + frozen_hash = digest(self.fixture.spec) + self.fixture.run_all() + result = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(result["evaluation"]["verdict"], "promotion_proposal") + self.assertEqual(result["experiment"]["project_path"], str(alias)) + self.assertEqual(result["evaluation"]["spec_sha256"], frozen_hash) + store = self.fixture.store() + try: + declared = store.connection.execute( + "SELECT spec_json,spec_sha256 FROM experiment_specs WHERE experiment_id=?", + (self.fixture.spec["experiment_id"],), + ).fetchone() + saved = store.connection.execute( + "SELECT spec_json,spec_sha256 FROM experiments WHERE experiment_id=?", + (self.fixture.spec["experiment_id"],), + ).fetchone() + self.assertEqual(tuple(saved), tuple(declared)) + self.assertEqual(saved["spec_sha256"], frozen_hash) + finally: + store.close() + + def test_reader_uses_prelaunch_snapshot_not_later_mutable_snapshot(self): + self.fixture.run_all() + run_id = self.fixture.runs[("hold-1", "candidate")] + store = self.fixture.store() + try: + original = store.run(run_id)["mutable_snapshot"] + changed = json.loads(original) + changed["task"]["goal"] = "Mutable continuation no longer describes the original trial." + changed["routing"]["roles"]["reviewer"]["selected"]["profile"]["model_id"] = "changed-model" + changed["experiment_assignment"]["spec_sha256"] = "0" * 64 + store.connection.execute( + "UPDATE runs SET mutable_snapshot=? WHERE id=?", (canonical_json(changed), run_id), + ) + try: + result = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(result["evaluation"]["verdict"], "promotion_proposal") + finally: + store.connection.execute("UPDATE runs SET mutable_snapshot=? WHERE id=?", (original, run_id)) + replay = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(replay, {**result, "replayed": True}) + finally: + store.close() + + def test_success_after_real_wrong_profile_fallback_cannot_credit_declared_arm(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "fallback", with_fallback=True) + self.addCleanup(fixture.close) + run_id = fixture.run_arm("eval-1", "candidate") + self.assertEqual(fixture.service.status(run_id)["state"], "succeeded") + store = fixture.store() + try: + attempts = store.connection.execute( + "SELECT profile_id,profile_index,output_metadata FROM attempts WHERE run_id=? ORDER BY profile_index", + (run_id,), + ).fetchall() + self.assertEqual([row["profile_id"] for row in attempts], ["candidate-fixture-fail", "profile-fallback"]) + self.assertEqual([row["profile_index"] for row in attempts], [0, 1]) + self.assertIsNotNone(json.loads(attempts[0]["output_metadata"])["failure"]) + self.assertEqual(len(store.outcomes_for_run(run_id)), 1) + with self.assertRaisesRegex(ContractError, "declared arm"): + fixture.service.policy_evaluate(fixture.spec) + self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM experiments").fetchone()[0], 0) + finally: + store.close() + + def test_missing_partner_is_visible_without_an_invented_pair(self): + self.fixture.run_all(skip=("hold-1", "control")) + evaluation = self.fixture.service.policy_evaluate(self.fixture.spec)["evaluation"] + self.assertEqual(evaluation["verdict"], "no_change") + self.assertEqual(evaluation["metrics"]["evaluation"]["available_pairs"], 1) + self.assertEqual(evaluation["metrics"]["held_out"]["available_pairs"], 0) + self.assertIn("insufficient_held_out_pairs", evaluation["reasons"]) + row = next(row for row in evaluation["cases"] if row["case_id"] == "hold-1") + self.assertEqual(row["status"], "missing") + self.assertEqual(row["missing"], ["control"]) + + def test_public_prelaunch_cancel_has_no_trial_exposure(self): + self.fixture.run_all(no_attempt=("hold-1", "candidate")) + cancelled_run = self.fixture.runs[("hold-1", "candidate")] + store = self.fixture.store() + try: + self.assertEqual(store.run(cancelled_run)["state"], "cancelled") + self.assertEqual(store.connection.execute( + "SELECT COUNT(*) FROM attempts WHERE run_id=?", (cancelled_run,), + ).fetchone()[0], 0) + self.assertEqual(len(store.outcomes_for_run(cancelled_run)), 1) + finally: + store.close() + evaluation = self.fixture.service.policy_evaluate(self.fixture.spec)["evaluation"] + self.assertEqual(evaluation["verdict"], "no_change") + self.assertEqual(evaluation["metrics"]["evaluation"]["available_pairs"], 1) + self.assertEqual(evaluation["metrics"]["held_out"]["available_pairs"], 0) + row = next(row for row in evaluation["cases"] if row["case_id"] == "hold-1") + self.assertEqual(row["status"], "missing") + self.assertIn("candidate", row["missing"]) + + def test_saved_tampering_is_rejected_even_when_partner_is_missing(self): + self.fixture.run_all(skip=("hold-1", "control")) + run_id = self.fixture.runs[("hold-1", "candidate")] + store = self.fixture.store() + try: + attempt = dict(store.connection.execute( + "SELECT * FROM attempts WHERE run_id=?", (run_id,), + ).fetchone()) + assignment = dict(store.connection.execute( + "SELECT * FROM experiment_assignments WHERE run_id=?", (run_id,), + ).fetchone()) + changed_assignment = json.loads(assignment["assignment_json"]) + changed_assignment["profile_id"] = "profile-a" + changed_snapshot = json.loads(assignment["snapshot_json"]) + changed_snapshot["task"]["goal"] = "A changed input after preparation." + mutations = [ + ("actual_profile", "attempts", "id", attempt["id"], + {"profile_id": "profile-a"}, {"profile_id": attempt["profile_id"]}), + ("actual_index", "attempts", "id", attempt["id"], + {"profile_index": 1}, {"profile_index": attempt["profile_index"]}), + ("actual_package", "attempts", "id", attempt["id"], + {"package_digest": "0" * 64}, {"package_digest": attempt["package_digest"]}), + ("saved_assignment", "experiment_assignments", "run_id", run_id, + {"assignment_json": canonical_json(changed_assignment), "assignment_sha256": digest(changed_assignment)}, + {"assignment_json": assignment["assignment_json"], "assignment_sha256": assignment["assignment_sha256"]}), + ("paired_input", "experiment_assignments", "run_id", run_id, + {"snapshot_json": canonical_json(changed_snapshot)}, {"snapshot_json": assignment["snapshot_json"]}), + ] + # SQL is intentionally limited to these negative corruptions. The + # positive runs above were prepared/executed/completed publicly. + for name, table, key, identity, changed, original in mutations: + with self.subTest(mutation=name): + columns = ",".join(f"{column}=?" for column in changed) + statement = f"UPDATE {table} SET {columns} WHERE {key}=?" + store.connection.execute(statement, [*changed.values(), identity]) + try: + with self.assertRaises(ContractError): + self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(store.connection.execute( + "SELECT COUNT(*) FROM experiments WHERE experiment_id=?", + (self.fixture.spec["experiment_id"],), + ).fetchone()[0], 0) + finally: + store.connection.execute(statement, [*original.values(), identity]) + finally: + store.close() + + def test_late_correction_invalidates_replay_without_rewriting_historical_evaluation(self): + self.fixture.run_all() + first = self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(first["evaluation"]["verdict"], "promotion_proposal") + store = self.fixture.store() + try: + original = dict(store.connection.execute( + "SELECT spec_json,spec_sha256,evaluation_json,evaluation_sha256,recorded_at " + "FROM experiments WHERE experiment_id=?", (self.fixture.spec["experiment_id"],), + ).fetchone()) + finally: + store.close() + correction = experimental_final("escaped-candidate-hold-1", "succeeded") + correction.update({ + "kind": "late_correction", "verdict": "escaped_defect", + "corrects_outcome_id": "candidate-hold-1", + "observed_at": datetime.now(timezone.utc).isoformat(), + "summary": "An escaped defect was found after the saved evaluation.", + }) + run_id = self.fixture.runs[("hold-1", "candidate")] + self.fixture.service.outcome_add(run_id, correction) + with self.assertRaises(ContractError): + self.fixture.service.policy_evaluate(self.fixture.spec) + store = self.fixture.store() + try: + saved = dict(store.connection.execute( + "SELECT spec_json,spec_sha256,evaluation_json,evaluation_sha256,recorded_at " + "FROM experiments WHERE experiment_id=?", (self.fixture.spec["experiment_id"],), + ).fetchone()) + self.assertEqual(saved, original) + self.assertEqual(len(store.outcomes_for_run(run_id)), 2) + finally: + store.close() + + +if __name__ == "__main__": + unittest.main() From 0d37764a3c2bb180d6ffd0e7933b1a9c0fff9567 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 13:16:32 -0700 Subject: [PATCH 140/197] WIP checkpoint: Jev private env-file setup and offline secret-safety tests (2026-10-01 13:16) --- .env.example | 4 + .gitignore | 5 + .../engineering-team/DECISION-CLASSIFIERS.md | 19 ++- docs/plans/engineering-team/RESUME.md | 21 +++- docs/plans/engineering-team/backlog.json | 13 ++- test/core/probes/jev_decision_eval.py | 71 +++++++++++- test/core/test_jev_decision_probe.py | 108 ++++++++++++++++++ 7 files changed, 230 insertions(+), 11 deletions(-) create mode 100644 .env.example diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..643ee9f --- /dev/null +++ b/.env.example @@ -0,0 +1,4 @@ +# Copy to .env, set permissions to 600, and fill the key locally. +# Never paste credentials into chat or commit them. +# The Jev probe reads this only with --env-file .env; runtime routing stays off. +TYPESAFE_API_KEY= diff --git a/.gitignore b/.gitignore index 73f1de5..dbdade6 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,11 @@ .claude/ .agent/ +# Local credentials; commit only the blank example. +.env +.env.* +!.env.example + # Dependencies node_modules/ __pycache__/ diff --git a/docs/plans/engineering-team/DECISION-CLASSIFIERS.md b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md index 3f0628d..15242fd 100644 --- a/docs/plans/engineering-team/DECISION-CLASSIFIERS.md +++ b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md @@ -175,11 +175,26 @@ latency and usage—not task text. ```bash python3 test/core/probes/jev_decision_eval.py -# Export TYPESAFE_API_KEY without placing its value in shell history, then run: -python3 test/core/probes/jev_decision_eval.py --execute \ +# Fill TYPESAFE_API_KEY in the Git-ignored local .env using your editor. +# For a fresh checkout, copy the blank .env.example to .env first. +chmod 600 .env +# This validates local configuration without calling the API or printing the key: +python3 test/core/probes/jev_decision_eval.py --env-file .env +# Live execution is a separate, explicitly selected step after pricing review: +python3 test/core/probes/jev_decision_eval.py --env-file .env --execute \ --output "$HOME/.devsquad/private-probes/jev-pilot-v1.json" ``` +The probe supports a single-line plain or quoted `TYPESAFE_API_KEY` assignment +and optional `export`; it ignores unrelated variables, rejects duplicate key +assignments, and never evaluates shell substitutions. The env file must be a +private regular UTF-8 file, at most 16 KiB, with no group/other permissions. +An already exported non-empty key takes precedence. Files are loaded only +when `--env-file` is supplied; this does not enable the runtime classifier or +pass the key to coding workers. A dry run reports only whether a key is present, +not whether TypeSafe has authenticated it. The local `.env` and `.env.*` files +are Git-ignored; `.env.example` contains no secret. + Do not copy the key or private receipt into Git. If the provider's current price makes one maximum-context request exceed $0.01, if returned usage breaches the cap, or if access requires purchasing a larger commitment, stop without retry diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 0018ae4..762f483 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -38,8 +38,25 @@ Gemini/Antigravity's September 29 live `squad_status` receipt proves read-only MCP observation of the same terminal-created saved run. It does not prove Gemini implementation/review worker execution or the updated installation. No extra Gemini login action is currently recorded; keep the normal session -signed in. Claude requires normal login; Grok's last verified authentication -was expired. The optional Jev key is separate from normal operation. +signed in. On October 1, a non-generating `claude auth status` check confirmed +`loggedIn=true` through normal Claude authentication. The user reports Grok +signed in as well; its installed non-generating diagnostics do not establish +authentication, so its bounded live operation remains unverified. No model +request was made for those checks. + +Local Jev `.env` setup is prepared: the root file has a blank +`TYPESAFE_API_KEY`, private mode 600 and Git ignore protection; the tracked +`.env.example` is blank. The probe now accepts explicit `--env-file .env`, +never executes shell text, preserves exported-key precedence and reports +only key presence during dry runs. All 14 focused probe tests pass, including +one mocked request and no-network dry runs. The 227-assertion Bash suite, +generated reference, JSON and whitespace checks pass; `.env` is untracked and +ignored while the blank example is trackable. No full core rerun was needed +for this isolated probe/setup change; R3b.1's full core gate remains pending. +The key must be filled locally; +no TypeSafe authentication/API request has run and the classifier remains off. +This setup does not close M6-D2 or change R3b.1's next action above. Recheck +current provider pricing before any separately selected live pilot. The earlier requested action was a review and execution plan for SOL. The October 1 planning pass inspected `b766f9e` / source `672e383` and updated diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index c7740bb..0903988 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -355,8 +355,17 @@ "depends_on": ["M6-D1"], "status": "blocked", "specification": "DECISION-CLASSIFIERS.md", - "evidence": [], - "blocker": "The TypeSafe console is at login and no TYPESAFE_API_KEY is available; one billable request, no retries, synthetic input and a $0.01 ceiling are prepared" + "evidence": [ + { + "kind": "local_env_setup", + "recorded_on": "2026-10-01", + "artifact": "../../../test/core/test_jev_decision_probe.py", + "command_or_action": "14 offline Jev probe tests, explicit --env-file dry run, private local .env and Git-ignore checks; 227 Bash assertions in 11 files, generated-reference, JSON and whitespace checks passed", + "outcome": "Explicit private env-file loading is verified without shell evaluation, key disclosure or a live API call. Local key field remains blank; authentication and the capped live pilot are still pending.", + "availability": "tracked_tests_and_blank_example" + } + ], + "blocker": "Private Git-ignored .env template and explicit probe loading are prepared, but TYPESAFE_API_KEY still needs local entry. No live TypeSafe authentication or pilot has run; recheck current pricing before the authorized one-request, no-retry, $0.01-capped synthetic pilot." }, { "id": "M6-D3", diff --git a/test/core/probes/jev_decision_eval.py b/test/core/probes/jev_decision_eval.py index 3a6c1dd..f04a371 100644 --- a/test/core/probes/jev_decision_eval.py +++ b/test/core/probes/jev_decision_eval.py @@ -2,8 +2,9 @@ """Run one bounded, synthetic Jev decision-classifier pilot. The probe is deliberately separate from the DevSquad runtime. It makes exactly -one billable request, performs no retries, accepts the API key only through the -environment, and writes a redacted result containing no task text. +one billable request, performs no retries, accepts the API key through the +environment or an explicitly selected private env file, and writes a redacted +result containing no task text. Dry run is the default and makes no API call. """ from __future__ import annotations @@ -15,7 +16,9 @@ import math import os from pathlib import Path +import re import ssl +import stat import sys import tempfile import time @@ -37,12 +40,62 @@ OFFICIAL_ENDPOINT = "https://api.typesafe.ai/v1/systemone" PINNED_MODEL = "jev-1.13.0" MAX_RESPONSE_BYTES = 1_048_576 +MAX_ENV_BYTES = 16_384 class ProbeError(RuntimeError): """A safe, user-facing probe failure.""" +def _key_value(value: str) -> str | None: + if any(character.isspace() or ord(character) < 32 or ord(character) == 127 + for character in value): + raise ProbeError("TYPESAFE_API_KEY must be a single non-whitespace value") + return value or None + + +def load_api_key(env_file: Path | None = None) -> str | None: + """Read only the Jev key; never source shell code or mutate process env.""" + exported = os.environ.get("TYPESAFE_API_KEY") + if exported: + return _key_value(exported) + if env_file is None: + return None + try: + descriptor = os.open(env_file, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + with os.fdopen(descriptor, "r", encoding="utf-8") as handle: + info = os.fstat(handle.fileno()) + if not stat.S_ISREG(info.st_mode): + raise ProbeError("Jev env file must be a regular file") + if info.st_mode & 0o077: + raise ProbeError("Jev env file permissions must be private (chmod 600)") + if info.st_size > MAX_ENV_BYTES: + raise ProbeError("Jev env file exceeded the size limit") + contents = handle.read(MAX_ENV_BYTES + 1) + except (OSError, UnicodeError): + raise ProbeError("Jev env file could not be read; use a private UTF-8 regular file") from None + if len(contents) > MAX_ENV_BYTES: + raise ProbeError("Jev env file exceeded the size limit") + key = None + found = False + for line in contents.splitlines(): + match = re.fullmatch(r"(?:export[ \t]+)?TYPESAFE_API_KEY[ \t]*=[ \t]*(.*)", line.strip()) + if match is None: + continue + if found: + raise ProbeError("Jev env file defines TYPESAFE_API_KEY more than once") + found = True + value = match[1].strip() + if value.startswith(("'", '"')): + if len(value) < 2 or value[-1] != value[0]: + raise ProbeError("Jev env file has an invalid quoted TYPESAFE_API_KEY") + value = value[1:-1] + else: + value = value.partition(" #")[0].rstrip() + key = _key_value(value) + return key + + def load_spec(path: Path) -> dict[str, Any]: with path.open(encoding="utf-8") as handle: spec = json.load(handle) @@ -271,8 +324,10 @@ def write_json(path: Path, value: dict[str, Any]) -> None: temporary.replace(path) -def execute(spec: dict[str, Any], request_body: dict[str, Any]) -> dict[str, Any]: - key = os.environ.get("TYPESAFE_API_KEY") +def execute( + spec: dict[str, Any], request_body: dict[str, Any], *, env_file: Path | None = None, +) -> dict[str, Any]: + key = load_api_key(env_file) if not key: raise ProbeError("TYPESAFE_API_KEY is not set") encoded = json.dumps(request_body, separators=(",", ":")).encode("utf-8") @@ -314,6 +369,10 @@ def main(argv: list[str] | None = None) -> int: parser.add_argument("--spec", type=Path, default=DEFAULT_SPEC) parser.add_argument("--execute", action="store_true") parser.add_argument("--output", type=Path) + parser.add_argument( + "--env-file", type=Path, + help="Read TYPESAFE_API_KEY from a private file; an exported key takes precedence", + ) args = parser.parse_args(argv) try: spec = load_spec(args.spec) @@ -343,13 +402,15 @@ def main(argv: list[str] | None = None) -> int: "data_class": spec["budget"]["data_class"], } if not args.execute: + if args.env_file is not None: + dry_run["api_key_configured"] = bool(load_api_key(args.env_file)) print(json.dumps(dry_run, indent=2, sort_keys=True)) return 0 if args.output is None: raise ProbeError("--output is required for a live run") if args.output.resolve().is_relative_to(ROOT): raise ProbeError("live output must remain outside the Git repository") - result = execute(spec, request_body) + result = execute(spec, request_body, env_file=args.env_file) write_json(args.output, result) print(json.dumps({**dry_run, "result": str(args.output)}, indent=2, sort_keys=True)) if result["pricing"]["estimated_cost_usd"] > max_cost: diff --git a/test/core/test_jev_decision_probe.py b/test/core/test_jev_decision_probe.py index fbd1731..2b469ca 100644 --- a/test/core/test_jev_decision_probe.py +++ b/test/core/test_jev_decision_probe.py @@ -1,12 +1,17 @@ from __future__ import annotations import importlib.util +import contextlib +import io import json import math +import os from pathlib import Path import subprocess import sys +import tempfile import unittest +from unittest import mock ROOT = Path(__file__).resolve().parents[2] @@ -124,5 +129,108 @@ def test_live_result_cannot_be_written_into_the_repository(self): self.assertIn("outside the Git repository", completed.stderr) +class JevEnvFileTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + self.env_file = Path(self.temporary.name) / ".env" + environment = mock.patch.dict(os.environ, {}, clear=True) + environment.start() + self.addCleanup(environment.stop) + + def write_env(self, text): + self.env_file.write_text(text, encoding="utf-8") + self.env_file.chmod(0o600) + + def test_explicit_env_file_supports_plain_and_quoted_key(self): + for declaration in ( + "TYPESAFE_API_KEY=fake-test-key", + "TYPESAFE_API_KEY='fake-test-key'", + 'export TYPESAFE_API_KEY="fake-test-key"', + ): + with self.subTest(declaration=declaration): + self.write_env("# Local credentials\nUNRELATED=ignored\n" + declaration + "\n") + self.assertEqual(jev.load_api_key(self.env_file), "fake-test-key") + self.assertNotIn("TYPESAFE_API_KEY", os.environ) + + def test_exported_key_wins_without_reading_env_file(self): + with mock.patch.dict(os.environ, {"TYPESAFE_API_KEY": "exported-test-key"}): + self.assertEqual(jev.load_api_key(self.env_file), "exported-test-key") + + def test_no_implicit_env_loading_and_blank_template_has_no_key(self): + self.write_env("TYPESAFE_API_KEY=local-test-key\n") + with mock.patch.object(jev.os, "open", side_effect=AssertionError("implicit read")): + self.assertIsNone(jev.load_api_key()) + self.write_env("TYPESAFE_API_KEY=\n") + self.assertIsNone(jev.load_api_key(self.env_file)) + + def test_values_are_literal_not_shell_expanded(self): + self.write_env("TYPESAFE_API_KEY='$UNDEFINED_KEY'\n") + self.assertEqual(jev.load_api_key(self.env_file), "$UNDEFINED_KEY") + + def test_malformed_and_duplicate_key_errors_never_echo_contents(self): + for contents in ( + "TYPESAFE_API_KEY='private-test-secret\n", + "TYPESAFE_API_KEY=private-test-secret with-spaces\n", + "TYPESAFE_API_KEY=private-test-secret\nTYPESAFE_API_KEY=second-key\n", + ): + with self.subTest(): + self.write_env(contents) + with self.assertRaises(jev.ProbeError) as caught: + jev.load_api_key(self.env_file) + self.assertNotIn("private-test-secret", str(caught.exception)) + + def test_missing_insecure_symlink_and_oversized_files_fail_safely(self): + with self.assertRaises(jev.ProbeError): + jev.load_api_key(self.env_file) + self.write_env("TYPESAFE_API_KEY=private-test-secret\n") + self.env_file.chmod(0o644) + with self.assertRaisesRegex(jev.ProbeError, "permissions"): + jev.load_api_key(self.env_file) + self.env_file.chmod(0o600) + link = Path(self.temporary.name) / "linked.env" + link.symlink_to(self.env_file) + with self.assertRaises(jev.ProbeError): + jev.load_api_key(link) + self.write_env("#" * (jev.MAX_ENV_BYTES + 1)) + with self.assertRaisesRegex(jev.ProbeError, "size"): + jev.load_api_key(self.env_file) + + def test_env_file_dry_run_only_reports_presence_and_never_calls_api(self): + for value, configured in (("", False), ("private-test-secret", True)): + with self.subTest(configured=configured): + self.write_env(f"TYPESAFE_API_KEY={value}\n") + stdout, stderr = io.StringIO(), io.StringIO() + with mock.patch.object(jev, "urlopen") as network, \ + contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + result = jev.main(["--env-file", str(self.env_file)]) + self.assertEqual(result, 0, stderr.getvalue()) + self.assertEqual(json.loads(stdout.getvalue())["api_key_configured"], configured) + self.assertNotIn("private-test-secret", stdout.getvalue() + stderr.getvalue()) + network.assert_not_called() + + def test_blank_key_blocks_execution_before_network(self): + self.write_env("TYPESAFE_API_KEY=\n") + spec = jev.load_spec(SPEC) + with mock.patch.object(jev, "urlopen") as network: + with self.assertRaisesRegex(jev.ProbeError, "not set"): + jev.execute(spec, jev.build_request(spec), env_file=self.env_file) + network.assert_not_called() + + def test_explicit_key_is_used_for_exactly_one_mocked_request(self): + self.write_env("TYPESAFE_API_KEY=private-test-secret\n") + spec = jev.load_spec(SPEC) + fixture = JevDecisionProbeTest() + fixture.setUp() + response = io.BytesIO(json.dumps(fixture.valid_response()).encode()) + with mock.patch.object(jev, "urlopen", return_value=response) as network: + result = jev.execute(spec, jev.build_request(spec), env_file=self.env_file) + network.assert_called_once() + request = network.call_args.args[0] + self.assertEqual(request.get_header("Authorization"), "Bearer private-test-secret") + self.assertNotIn("private-test-secret", json.dumps(result)) + self.assertNotIn("TYPESAFE_API_KEY", os.environ) + + if __name__ == "__main__": unittest.main() From bd0cf5c2fd0cc9d924676072d184f8f0693c0fea Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 13:42:47 -0700 Subject: [PATCH 141/197] WIP checkpoint: Record one-request Jev pilot, measured tier disagreements and keep-off decision (2026-10-01 13:42) --- .../engineering-team/DECISION-CLASSIFIERS.md | 17 +++- docs/plans/engineering-team/M6-STATUS.md | 17 ++-- docs/plans/engineering-team/RESUME.md | 55 ++++++++----- docs/plans/engineering-team/START-HERE.md | 6 +- docs/plans/engineering-team/backlog.json | 16 +++- .../evidence/M6-D2-jev-pilot-2026-10-01.json | 77 +++++++++++++++++++ 6 files changed, 154 insertions(+), 34 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/M6-D2-jev-pilot-2026-10-01.json diff --git a/docs/plans/engineering-team/DECISION-CLASSIFIERS.md b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md index 15242fd..6dc81f2 100644 --- a/docs/plans/engineering-team/DECISION-CLASSIFIERS.md +++ b/docs/plans/engineering-team/DECISION-CLASSIFIERS.md @@ -24,9 +24,20 @@ number of saved subscription windows has been measured. The initial evaluation inspected public documentation, benchmark reports and pinned Laya source. It did **not** install weights or run Laya inference. The tracked [Jev pilot specification](experiments/jev-pilot-v1.json) and validated -[probe](../../../test/core/probes/jev_decision_eval.py) are now ready, but no -TypeSafe request has run because this environment has no `TYPESAFE_API_KEY` and -the console is at its login screen. The live result remains pending. +[probe](../../../test/core/probes/jev_decision_eval.py) subsequently completed +their one-request live smoke on October 1. See the +[redacted receipt summary](evidence/M6-D2-jev-pilot-2026-10-01.json): exact pinned +model, 5,373 reported input tokens, estimated $0.00022567, 8/8 task-family and +skill labels, but 6/8 execution-tier labels. The two tier disagreements remain +failures against the frozen expectations; the classifier is still off. + +This closes M6-D2's synthetic mechanics/access measurement only. It does not +prove production quality, calibration, speedup or adoption. The cost/access +Laya trigger did not fire; this smoke has no frozen numeric production accuracy +threshold from which to claim a quality-triggered switch. Broader evaluation +needs predeclared data, held-out gates and separately authorized resources. +The existing one-request allowance is spent: do not rerun the live command +below without new authorization. ## What was verified diff --git a/docs/plans/engineering-team/M6-STATUS.md b/docs/plans/engineering-team/M6-STATUS.md index 80cae05..6941d13 100644 --- a/docs/plans/engineering-team/M6-STATUS.md +++ b/docs/plans/engineering-team/M6-STATUS.md @@ -5,8 +5,9 @@ The component tests below remain useful historical evidence, but the September 29 review found reused held-out evidence (F3), normal routing that bypasses lifecycle bindings (F4), and missing public outcome/experiment, catalog and quota connections (G1/G2). Execute R3–R5 in -[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The Jev key blocks only its -separate measurement; it does not block this engineering work. +[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md). The separate one-request Jev +smoke completed on October 1; it does not close this engineering work or enable +runtime routing. Its preserved two execution-tier disagreements remain visible. R3a now rejects reused outcomes and adds v2 concrete-profile/paired-input contracts, a stable corpus identity distinct from runtime context, strict @@ -33,7 +34,7 @@ do not treat legacy helper fixtures as public proof. | Model lifecycle | Templates, qualification budgets, reviewed/guarded-auto promotion, compare-and-swap bindings, new-run-only effects and rollback receipts | components through `ca4ee73`; qualification integrity and normal alias routing reopened under R3/R4 | | Catalog drift and unavailable incumbent | Complete catalog drift scopes revalidation; added models stay unqualified; removed incumbents roll back only to a prior proven/qualified binding or block | component tests at `398ae6a` / `ca4ee73`; production discovery connection pending R4 | | Decision helper M6-D1 | Default-off typed contract, fake adapter, cache/accounting and authority/integrity tests | verified at `87fa9cf` | -| Jev M6-D2 | One capped synthetic request with exact model/usage/latency/cost receipt | blocked on `TYPESAFE_API_KEY` | +| Jev M6-D2 | One capped synthetic request with exact model/usage/latency/cost receipt | smoke complete October 1; 8/8 family and skill labels, 6/8 tiers; runtime remains off | | Laya M6-D3 | Triggered pinned local comparison and measured keep-off/adopt decision | pending; run only if the declared Jev trigger fires | ## Capacity checkpoint @@ -148,7 +149,9 @@ with 2 optional-SDK skips** and **220/220 Bash assertions**. ## Exact next slice -Begin R3's failing repeated-outcome regression, then wire R4/R5 through public -saved runs. Keep the one-request Jev M6-D2 gate blocked until -`TYPESAFE_API_KEY` is supplied. Do not install or run Laya unless its declared -trigger fires. A synthetic pilot does not prove production routing quality. +Finish R3b.1 from the current partial saved-run reader, then shared eligibility +and historical compatibility before wiring R4/R5 through public saved runs. +Jev M6-D2's one-request allowance is spent; preserve its +[receipt summary](evidence/M6-D2-jev-pilot-2026-10-01.json) and do not repeat it. +Keep runtime classification off. Do not install or run Laya unless its declared +trigger is established. A synthetic pilot does not prove production routing quality. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 762f483..25707e4 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -44,8 +44,8 @@ signed in as well; its installed non-generating diagnostics do not establish authentication, so its bounded live operation remains unverified. No model request was made for those checks. -Local Jev `.env` setup is prepared: the root file has a blank -`TYPESAFE_API_KEY`, private mode 600 and Git ignore protection; the tracked +Local Jev `.env` setup is complete: the user filled `TYPESAFE_API_KEY` locally, +and the file retains private mode 600 and Git ignore protection; the tracked `.env.example` is blank. The probe now accepts explicit `--env-file .env`, never executes shell text, preserves exported-key precedence and reports only key presence during dry runs. All 14 focused probe tests pass, including @@ -53,10 +53,24 @@ one mocked request and no-network dry runs. The 227-assertion Bash suite, generated reference, JSON and whitespace checks pass; `.env` is untracked and ignored while the blank example is trackable. No full core rerun was needed for this isolated probe/setup change; R3b.1's full core gate remains pending. -The key must be filled locally; -no TypeSafe authentication/API request has run and the classifier remains off. -This setup does not close M6-D2 or change R3b.1's next action above. Recheck -current provider pricing before any separately selected live pilot. +The October 1 live M6-D2 pilot has now completed exactly one request and no +retries after current official pricing/API revalidation. Jev reported the +pinned `jev-1.13.0`, 5,373 input and 1,762 output tokens, and 187 ms for this +single sample. Estimated input charge is $0.00022567, below the $0.01 ceiling; +the invoice was not inspected. Task-family and skill labels each matched 8/8; +execution-tier matched 6/8. T07/T08 were lower than their frozen expected tiers, +including a wrong T07 label with 0.92 confidence. Do not change gold labels or +treat confidence as proven correctness. + +M6-D2 closes only the bounded synthetic smoke. Runtime classification remains +off; no measured routing benefit or quality/adoption gate is claimed. The +cost/access Laya trigger did not fire, and this smoke had no numeric production +quality threshold. No Laya setup or further hosted request is authorized by +this result. **Do not repeat the pilot:** its one-request allowance is spent. +The private receipt is outside Git; its hash and portable summary are in +[Jev pilot evidence](evidence/M6-D2-jev-pilot-2026-10-01.json). R3b.1's exact +next action above is unchanged; later shadow/adoption comparisons require a +predeclared corpus, held-out gates and separately approved budget. The earlier requested action was a review and execution plan for SOL. The October 1 planning pass inspected `b766f9e` / source `672e383` and updated @@ -342,8 +356,9 @@ the review correction above governs current completion and next work. Malformed, unknown-ID, NaN, pin, permission/quality, drift, cancellation and crash/resume cases fail closed. The frozen synthetic baseline explicitly keeps runtime adoption off. The gate is 291 core tests with 2 optional-SDK - skips and 220 Bash assertions. M6-D2 remains blocked only on - `TYPESAFE_API_KEY`; Laya remains conditional on its declared trigger. + skips and 220 Bash assertions. M6-D2 was then blocked on the key; the October 1 + one-request smoke above resolves that measurement only. Laya remains + conditional on its declared trigger and runtime guidance remains off. - M7 packaging, normal task entry and the currently available live surfaces are verified through `b1d52ad`. The immutable standalone installer works without Claude, performs offline exact-lock MCP installation with `pip check`, emits @@ -382,14 +397,13 @@ the review correction above governs current completion and next work. Jev pilot, and local Laya fallback plus measured adoption. It prioritizes routing hints, skill/tool shortlists and context ranking, followed by failure triage, review attention and outcome labels. No weights/inference/API spending or - runtime routing changes occurred. The user has authorized one Jev request + runtime routing changes occurred during that original planning pass. The user authorized one Jev request using only the synthetic fixture, no retries and at most $0.01. The tracked fixture/probe checkpoint is committed at `70e59cb` and five focused offline tests are ready; the complete offline gate is 231 core tests discovered (suite OK, 2 optional SDK skips) and 220 - Bash assertions. The live call is - blocked because the TypeSafe console is at login and no `TYPESAFE_API_KEY` - exists. Classifier suggestions never become permission/acceptance authority. + Bash assertions. The live call was blocked until the October 1 key setup; + it has now run once as recorded above. Classifier suggestions never become permission/acceptance authority. Do not wait for this probe to execute the review repairs starting at R1. ## Completed and preserved @@ -454,13 +468,16 @@ provider paths must not be advertised as verified. compatibility, then execute R4–R6 in dependency order. Preserve R1/R2, explicit check `output_paths` contract and historical receipts. Do not rewrite the architecture or reset completed work. -3. Keep Claude and Grok live probes paused until normal login is restored. - Afterward, run the M4 Claude handoff, the M5 installed Claude-to-Codex - delivery and one bounded Grok operation, retaining only redacted evidence. -4. Keep M6 decision guidance off by default. Once `TYPESAFE_API_KEY` is - supplied, run the prepared one-request synthetic Jev pilot immediately with - no retry and the $0.01 ceiling. Install/run Laya only if the predeclared Jev - cost/access/quality trigger fires. +3. Claude login is now confirmed; the user reports Grok signed in. After the + repaired installation is safely refreshed under R8, run the M4 real Claude + handoff, the M5 installed Claude-to-Codex delivery and one bounded Grok + operation, retaining only redacted evidence. Authentication is not itself + a supported-operation receipt. +4. Keep M6 decision guidance off. The one-request Jev pilot is complete and + must not be repeated under its spent authorization. Preserve its tier + disagreements and frozen labels; a broader shadow/adoption comparison needs + predeclared gates and a separate budget. Install/run Laya only if the + declared cost/access/quality trigger is established. 5. Complete the separately gated C1 extension under R7 and audit installed/live closure under R8. C1 is required in the full assignment even though it does not reopen M7. Record each blocked subgate without pausing unrelated work. diff --git a/docs/plans/engineering-team/START-HERE.md b/docs/plans/engineering-team/START-HERE.md index 580999a..a8128db 100644 --- a/docs/plans/engineering-team/START-HERE.md +++ b/docs/plans/engineering-team/START-HERE.md @@ -40,8 +40,10 @@ The September 26 [Jev/Laya evaluation and decision-helper amendment](DECISION-CL adds optional M6 experiments for routing hints, skill selection and context ranking, with further bounded uses prioritized. It changes no runtime defaults and does not delay M5. The only current hosted authorization is the explicitly -capped one-request, $0.01 synthetic Jev pilot; no purchase, retry or private -task upload is authorized. +capped one-request, $0.01 synthetic Jev pilot; it completed on October 1 and +that allowance is now spent. See the [receipt summary](evidence/M6-D2-jev-pilot-2026-10-01.json). +No repeat request, purchase, retry or private task upload is authorized; runtime +classification remains off. ## Copyable execution brief diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 0903988..dd1ac65 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -353,7 +353,7 @@ "id": "M6-D2", "title": "Capped Jev synthetic pilot for task/profile and skill hints", "depends_on": ["M6-D1"], - "status": "blocked", + "status": "complete", "specification": "DECISION-CLASSIFIERS.md", "evidence": [ { @@ -363,9 +363,19 @@ "command_or_action": "14 offline Jev probe tests, explicit --env-file dry run, private local .env and Git-ignore checks; 227 Bash assertions in 11 files, generated-reference, JSON and whitespace checks passed", "outcome": "Explicit private env-file loading is verified without shell evaluation, key disclosure or a live API call. Local key field remains blank; authentication and the capped live pilot are still pending.", "availability": "tracked_tests_and_blank_example" + }, + { + "kind": "live_synthetic_smoke", + "revision": "0d37764", + "recorded_at": "2026-10-01T20:35:46.511565+00:00", + "artifact": "evidence/M6-D2-jev-pilot-2026-10-01.json", + "command_or_action": "One pinned Jev 1.13 API request after official pricing/API recheck; no retries; private env-file and redacted receipt outside Git", + "outcome": "Typed authenticated smoke completed: 5373 input tokens, 1762 output tokens, 187 ms and estimated $0.00022567. Task-family and skill labels matched 8/8 each; execution-tier matched 6/8. This closes the single-request smoke only, not production quality or routing adoption. Runtime remains off and the one-request allowance is spent.", + "availability": "portable_redacted_with_private_receipt_hash" } ], - "blocker": "Private Git-ignored .env template and explicit probe loading are prepared, but TYPESAFE_API_KEY still needs local entry. No live TypeSafe authentication or pilot has run; recheck current pricing before the authorized one-request, no-retry, $0.01-capped synthetic pilot." + "checkpoint": "One-request smoke complete; preserved two tier disagreements, exact model/usage/cost evidence and private receipt hash. No automatic routing change or additional hosted request authorized.", + "blocker": null }, { "id": "M6-D3", @@ -433,7 +443,7 @@ "availability": "tracked_fixture_and_tests" } ], - "blocker": "Independent R3-R5 evidence, routing, learning and observation integration remains. Only the separate single-request Jev measurement requires TYPESAFE_API_KEY; Laya remains conditional." + "blocker": "Independent R3-R5 evidence, routing, learning and observation integration remains. The single-request Jev smoke is complete but does not prove production quality; runtime guidance remains off and Laya remains conditional." }, { "id": "M7", diff --git a/docs/plans/engineering-team/evidence/M6-D2-jev-pilot-2026-10-01.json b/docs/plans/engineering-team/evidence/M6-D2-jev-pilot-2026-10-01.json new file mode 100644 index 0000000..cac3c82 --- /dev/null +++ b/docs/plans/engineering-team/evidence/M6-D2-jev-pilot-2026-10-01.json @@ -0,0 +1,77 @@ +{ + "schema_version": 1, + "work_package": "M6-D2", + "status": "complete_smoke_only", + "executed_at": "2026-10-01T20:35:46.511565+00:00", + "implementation_revision": "0d37764", + "experiment_id": "jev-devsquad-routing-pilot-v1", + "spec_sha256": "473aed2c431bd1ce8218124a6fe3fd3139fdf869468a72cbc8cf184c428d9822", + "probe_sha256": "cf665296d40d2c7b556efbfd8e8cc619e583e92f70715ee2c8eccfbecd80765b", + "request_sha256": "ae5d7f39d64cd35cd8e6c11ad78a8b00f4264755d8fe1cd557c64d80980e32e7", + "command": "python3 test/core/probes/jev_decision_eval.py --env-file .env --execute --output /Users/Dikshant/.devsquad/private-probes/jev-pilot-v1-20261001T203515Z.json", + "result": { + "exit_code": 0, + "client_request_count": 1, + "retries": 0, + "requested_model": "jev-1.13.0", + "observed_model": "jev-1.13.0", + "elapsed_ms": 187, + "cases": 8, + "questions": 24, + "input_tokens": 5373, + "output_tokens": 1762, + "task_family": {"correct": 8, "total": 8}, + "specialist_skill": {"correct": 8, "total": 8}, + "execution_tier": {"correct": 6, "total": 8} + }, + "pricing": { + "rechecked_on": "2026-10-01", + "usd_per_million_input_tokens": 0.042, + "output_tokens_charge": "free per current published pricing", + "documented_max_request_cost_usd": 0.002688, + "estimated_cost_usd_from_reported_usage": 0.00022567, + "authorized_ceiling_usd": 0.01, + "invoice_verified": false, + "source": "https://docs.typesafe.ai/models" + }, + "api_conformance": { + "source": "https://docs.typesafe.ai/api", + "rechecked_on": "2026-10-01", + "validated": "Pinned model, exact answer IDs, typed choices, finite probability distributions, confidence bounds and reported usage" + }, + "tier_disagreements": [ + { + "case_id": "T07", + "expected": "frontier_analysis", + "choice": "economy_read", + "confidence": 0.92 + }, + { + "case_id": "T08", + "expected": "frontier_write", + "choice": "standard_write", + "confidence": 0.23 + } + ], + "private_receipt": { + "path": "/Users/Dikshant/.devsquad/private-probes/jev-pilot-v1-20261001T203515Z.json", + "sha256": "fcf9e71f16e93f0b96cbf162d23ed92909c52cb3792e4b10a0c1c2a1f72e31c6", + "permissions": "600", + "contents": "Redacted scored synthetic result; no API key or task text" + }, + "checkpoint_verification": { + "bash": "11 files, 227 assertions passed", + "json_generated_reference_whitespace": "passed", + "credential_protection": "Local .env remains ignored and untracked; no key stored in this evidence", + "source_changes": "Documentation/evidence only; no full Python rerun. The probe's unchanged source retains its 14 passing focused tests; R3b.1 full integration remains pending." + }, + "decision": { + "runtime_mode": "off", + "active_policy_changed": false, + "scope": "One authenticated typed synthetic smoke, not production quality or speedup proof", + "cost_or_access_laya_trigger": false, + "quality_gate": "The smoke has no frozen numeric adoption threshold; two tier disagreements do not establish a passed production gate or automatically authorize a new comparison", + "additional_hosted_requests_authorized": 0, + "next_action": "Finish R3-R6 runtime repairs; predeclare corpus, held-out thresholds and budget before separately authorized shadow/adoption work. Do not repeat this pilot or install Laya without its declared trigger." + } +} From a4a87fd32a663227493932626c0a11bb862f21b6 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 13:58:38 -0700 Subject: [PATCH 142/197] WIP checkpoint: Repair and regress honest failed-review experiment evidence (2026-10-01 13:58) --- docs/plans/engineering-team/RESUME.md | 25 +++-- docs/plans/engineering-team/backlog.json | 6 +- .../R3b1-failure-path-2026-10-01.json | 28 ++++++ .../core/src/devsquad/experiment_evidence.py | 34 ++++++- test/core/experiment_runtime_fixture.py | 36 +++---- test/core/test_experiment_saved_runs.py | 94 +++++++++++++++++++ 6 files changed, 193 insertions(+), 30 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 25707e4..2b04b79 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -25,12 +25,21 @@ No full Python integration gate, installed refresh or provider call has run for this reader slice. The independent follow-up review hit its usage limit; do not claim an independent reader audit passed. -**Exact next action:** reproduce the reader's handling of an honest terminal -failed reviewer whose output metadata has no `failure` key. The present -successful-review stdout decoder is guarded by that key and may incorrectly -decode failed non-JSON output. This is an open inspection concern, not yet a -reproduced/fixed regression. Resolve it and run the complete core gate before -accepting R3b.1. Then finish R3b.2's shared current-evidence eligibility and +The terminal-failure concern is now reproduced and repaired. A real offline +failed reviewer exited with empty stdout and captured stderr, without a +`failure` metadata key; the old reader rejected it as invalid JSON. The reader +now requires either imported successful review evidence (still strictly +decoded) or a hash-verified terminal receipt with the exact failed/cancelled +attempt, role and frozen profile. A forged failure label cannot hide missing +successful evidence. Receipt run/profile/status/list corruptions fail closed. +The 34-test focused regression set passed in 42.301 seconds; six additional +success-integrity and saved-evidence tests passed in 9.803 seconds. The Bash +gate passed all 227 assertions. Source fingerprints and red/green results are +saved in [failure-path evidence](evidence/R3b1-failure-path-2026-10-01.json). + +**Exact next action:** run the complete core gate on the checkpointed reader +source before accepting R3b.1; no test is running at this checkpoint. Then +finish R3b.2's shared current-evidence eligibility and explicit append-only evaluation/review revisions; legacy/public compatibility remains R3c and the production paired-trial controller remains R5. @@ -458,8 +467,8 @@ provider paths must not be advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Finish R3b.1 from the preserved partial reader: reproduce/repair honest - terminal failed-output handling and run the full unchanged-source gate. +2. Finish R3b.1 from the preserved reader: honest terminal failed-output + handling is repaired; run the full unchanged-source gate. Then implement R3b.2 shared current-evidence eligibility and append-only evaluation/review revisions. Validate every relevant attempt, including fallbacks/repairs. Finish diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index dd1ac65..207f019 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -19,9 +19,9 @@ "recorded_on": "2026-10-01", "baseline_revision": "a22b949", "status": "partial", - "artifact": "evidence/R3b1-reader-partial-2026-10-01.json", + "artifact": "evidence/R3b1-failure-path-2026-10-01.json", "scope": "Authoritative v2 saved-run reader, replay invalidation, prelaunch controlled-input guard and public offline fixture; no install or provider calls", - "next_action": "Reproduce and repair honest terminal-failure stdout decoding, run full core integration, then R3b.2 shared current-evidence eligibility and append-only revisions", + "next_action": "Terminal-failure decoding reproduced and repaired; run full core integration, then R3b.2 shared current-evidence eligibility and append-only revisions", "limitations": "Full integration and independent reader audit not complete; production paired-trial controller remains R5; native Gemini worker and refreshed installed proof remain unverified" }, "planning_checkpoint": { @@ -44,7 +44,7 @@ "review_work_packages": [ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, - {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3b1-reader-partial-2026-10-01.json", "checkpoint": "R3a remains verified at 672e383 with its complete 402-test gate. R3b.1 now has a partial public v2 saved-run reader, changed-evidence replay rejection and controlled-input launch guard. Imported-profile mismatch was reproduced and repaired. Honest terminal-failure decoding is an open inspection concern; full integration and independent reader audit are not complete. Shared eligibility and append-only revisions remain R3b.2; legacy/public compatibility remains R3c. R3 is not closed."}, + {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3b1-failure-path-2026-10-01.json", "checkpoint": "R3a remains verified at 672e383 with its complete 402-test gate. R3b.1 has a public v2 saved-run reader, changed-evidence replay rejection and controlled-input launch guard. Imported-profile mismatch and honest terminal-failure decoding were reproduced and repaired. New failure-receipt and missing-success regressions pass; full integration and independent reader audit remain incomplete. Shared eligibility and append-only revisions remain R3b.2; legacy/public compatibility remains R3c. R3 is not closed."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, {"id": "R6", "title": "Normal terminal experience and readiness", "status": "pending", "milestones": ["M5", "M7"], "depends_on": ["R1", "R2", "R4", "R5"], "items": ["G3", "G4"]}, diff --git a/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json new file mode 100644 index 0000000..40c88fc --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json @@ -0,0 +1,28 @@ +{ + "schema_version": 1, + "work_package": "R3b.1", + "status": "failure_path_repaired_full_gate_pending", + "recorded_at": "2026-10-01T20:57:36Z", + "baseline_revision": "bd0cf5c", + "source_sha256": { + "plugin/core/src/devsquad/experiment_evidence.py": "38a11bca28bcde0835dee5f5aaaef450e5bed0e951b57edeec88063d4ee558ee", + "test/core/experiment_runtime_fixture.py": "08a8cf77c34052fc23fd782041d0d272110fbc1e3fa1db6bccaad8903005eaaf", + "test/core/test_experiment_saved_runs.py": "7652cae8b40d512c8019c58edc856537f34c27d06120c974040bd4f9953558dc" + }, + "verification": { + "red": "Real terminal failed worker regression errored in strict stdout JSON decoding; 1 test in 4.198 seconds, exit 1", + "initial_green_attempt": "Decoder repair reached evaluation, then the test itself raised KeyError on the wrong case-row key; 1 test in 4.297 seconds, exit 1, not counted as a pass", + "failure_and_negative_receipts": "2 tests passed in 5.363 seconds; valid failed exposure retained, malformed or mismatched hash-valid failure receipts rejected", + "focused_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_experiment_saved_runs test_experiment_provenance test_learning test_lifecycle -v", + "focused_result": "34 tests passed in 42.301 seconds, exit 0", + "integrity_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_experiment_saved_runs.ExperimentSavedRunsTest.test_success_cannot_hide_missing_review_behind_failure_metadata test_experiment_evidence_integrity -v", + "integrity_result": "6 tests passed in 9.803 seconds, exit 0", + "bash": "bash test/run.sh: 227 assertions in 11 files passed", + "generated_reference": "python3 scripts/generate-core-reference.py --check passed", + "full_core": "Pending; run on this checkpointed source", + "independent_review": "Not completed; no independent audit claimed" + }, + "fixture_boundary": "Real offline workers and public terminal completion; test-only predeclared-assignment seam, manual outcome addition and synthetic host dispositions. Not the R5 production trial controller or native quality proof.", + "external_actions": "No provider request, installation refresh, purchase, reset, global setting change or push", + "next_action": "Full core regression gate, then R3b.2 shared current-evidence eligibility and explicit append-only evaluation/review revisions" +} diff --git a/plugin/core/src/devsquad/experiment_evidence.py b/plugin/core/src/devsquad/experiment_evidence.py index e61e88c..b434719 100644 --- a/plugin/core/src/devsquad/experiment_evidence.py +++ b/plugin/core/src/devsquad/experiment_evidence.py @@ -215,6 +215,7 @@ def _attempts( incomplete = True # These immutable artifacts retain imported observed identity and # native usage; failure diagnostics are retained in output_metadata. + review_imported = False for prefix in ("review", "implementation", "lead"): artifact_row = connection.execute( "SELECT id FROM artifacts WHERE run_id=? AND name=?", @@ -228,11 +229,38 @@ def _attempts( or canonical_json(evidence.get("selected_profile")) != canonical_json(selected)): raise ContractError("experiment imported execution differs from its frozen attempt profile") artifacts.append(artifact) + review_imported = review_imported or prefix == "review" if (role == "reviewer" and snapshot["task"]["workflow"] == "branch-review" - and captures and metadata.get("failure") is None): - from .workflows import validate_branch_review_evidence + and captures): + if review_imported: + from .workflows import validate_branch_review_evidence - validate_branch_review_evidence(strict_json(captures["stdout"]), snapshot) + validate_branch_review_evidence(strict_json(captures["stdout"]), snapshot) + else: + # Terminal failed/cancelled workers do not publish successful + # review evidence. Bind their opaque output to the hashed + # early-terminal receipt, not an absent metadata.failure key. + receipt_row = connection.execute( + "SELECT id FROM artifacts WHERE run_id=? AND name='result-receipt.json'", + (run["id"],), + ).fetchone() + if receipt_row is None: + raise ContractError("experiment reviewer has no imported review or failure receipt") + artifact, content = _artifact(connection, receipt_row["id"], run["id"]) + receipt = _object(content, "terminal failure receipt") + receipt_attempts = receipt.get("attempts") + if not isinstance(receipt_attempts, list): + raise ContractError("experiment failed reviewer receipt attempts are invalid") + projections = [item for item in receipt_attempts + if isinstance(item, dict) and item.get("id") == row["id"]] + if (receipt.get("run_id") != run["id"] + or receipt.get("state") != run["state"] + or len(projections) != 1 + or projections[0].get("status") not in {"failed", "cancelled"} + or projections[0].get("role") != role + or canonical_json(projections[0].get("selected_profile")) != canonical_json(selected)): + raise ContractError("experiment failed reviewer receipt is inconsistent") + artifacts.append(artifact) history.append({ **{key: row[key] for key in ( "id", "run_id", "project_id", "role", "status", "profile_id", diff --git a/test/core/experiment_runtime_fixture.py b/test/core/experiment_runtime_fixture.py index 2cd097c..8340ffa 100644 --- a/test/core/experiment_runtime_fixture.py +++ b/test/core/experiment_runtime_fixture.py @@ -27,10 +27,11 @@ class ExperimentRuntimeFixture: - def __init__(self, root: Path, *, with_fallback=False): + def __init__(self, root: Path, *, with_fallback=False, fail_candidate=False): self.root = root.resolve() self.root.mkdir(parents=True, exist_ok=True) self.with_fallback = with_fallback + self.fail_candidate = fail_candidate self.repo = self.root / "repo" self.service = Service(self.root / "runtime") self.runs = {} @@ -52,7 +53,7 @@ def __init__(self, root: Path, *, with_fallback=False): arm: profile(f"profile-{letter}", f"model-{letter}") for arm, letter in (("control", "a"), ("candidate", "b")) } - if with_fallback: + if with_fallback or fail_candidate: # The real offline worker deliberately errors on this suffix. self.profiles["candidate"]["id"] = "candidate-fixture-fail" self.registry = { @@ -176,20 +177,23 @@ def predeclared_assignment(*args, **kwargs): verdict = "cancelled" else: waiting = self.wait(run_id) - if waiting["state"] != "awaiting_host": - raise AssertionError(f"public fixture worker failed: {waiting}") - claimed = self.service.handoff_claim(run_id, waiting["version"], "experiment-fixture-host") - packet = claimed["handoff"]["packet"] - body = { - "schema_version": 1, "submission_id": f"finish-{arm}-{case_id}", - "disposition": "accept" if arm == "candidate" else "reject", - "reason": "Predeclared offline fixture disposition.", - "evidence_refs": [{"artifact_id": reference["artifact_id"], "sha256": reference["sha256"]} for reference in packet["artifacts"]], - } - completed = self.service.handoff_complete(run_id, claimed["claim"], {**body, "submission_hash": request_hash(body)}) - verdict = "succeeded" if arm == "candidate" else "failed" - if completed["state"] != verdict: - raise AssertionError(f"public fixture disposition failed: {completed}") + if waiting["state"] == "failed" and self.fail_candidate and arm == "candidate": + verdict = "failed" + else: + if waiting["state"] != "awaiting_host": + raise AssertionError(f"public fixture worker failed: {waiting}") + claimed = self.service.handoff_claim(run_id, waiting["version"], "experiment-fixture-host") + packet = claimed["handoff"]["packet"] + body = { + "schema_version": 1, "submission_id": f"finish-{arm}-{case_id}", + "disposition": "accept" if arm == "candidate" else "reject", + "reason": "Predeclared offline fixture disposition.", + "evidence_refs": [{"artifact_id": reference["artifact_id"], "sha256": reference["sha256"]} for reference in packet["artifacts"]], + } + completed = self.service.handoff_complete(run_id, claimed["claim"], {**body, "submission_hash": request_hash(body)}) + verdict = "succeeded" if arm == "candidate" else "failed" + if completed["state"] != verdict: + raise AssertionError(f"public fixture disposition failed: {completed}") outcome = experimental_final(f"{arm}-{case_id}", verdict) outcome["observed_at"] = datetime.now(timezone.utc).isoformat() outcome["evidence_refs"] = ["receipt.json", "result-receipt.json"] diff --git a/test/core/test_experiment_saved_runs.py b/test/core/test_experiment_saved_runs.py index a561b4a..8aaf22d 100644 --- a/test/core/test_experiment_saved_runs.py +++ b/test/core/test_experiment_saved_runs.py @@ -69,6 +69,100 @@ def test_project_symlink_alias_preserves_the_predeclared_spec_hash(self): finally: store.close() + def test_real_terminal_worker_failure_is_retained_as_failed_exposure(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "terminal-failure", fail_candidate=True) + self.addCleanup(fixture.close) + fixture.run_all() + store = fixture.store() + try: + for case_id in ("eval-1", "hold-1"): + run_id = fixture.runs[(case_id, "candidate")] + self.assertEqual(store.run(run_id)["state"], "failed") + attempt = store.connection.execute( + "SELECT * FROM attempts WHERE run_id=?", (run_id,), + ).fetchone() + self.assertEqual(attempt["status"], "finished") + metadata = json.loads(attempt["output_metadata"]) + self.assertNotIn("failure", metadata) + self.assertEqual(metadata["stdout"]["captured_bytes"], 0) + self.assertGreater(metadata["stderr"]["captured_bytes"], 0) + finally: + store.close() + evaluation = fixture.service.policy_evaluate(fixture.spec)["evaluation"] + self.assertEqual(evaluation["verdict"], "no_change") + for split in ("evaluation", "held_out"): + self.assertEqual(evaluation["metrics"][split]["available_pairs"], 1) + self.assertEqual(evaluation["metrics"][split]["candidate_success_rate"], 0.0) + self.assertTrue(all(case["candidate_verdict"] == "failed" for case in evaluation["cases"])) + + def test_failed_receipt_requires_matching_terminal_attempt_even_with_valid_hash(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "failure-receipt", fail_candidate=True) + self.addCleanup(fixture.close) + run_id = fixture.run_arm("eval-1", "candidate") + store = fixture.store() + try: + artifact = dict(store.connection.execute( + "SELECT * FROM artifacts WHERE run_id=? AND name='result-receipt.json'", (run_id,), + ).fetchone()) + path = Path(artifact["path"]) + original = path.read_bytes() + receipt = json.loads(original) + changed_profile = json.loads(original) + changed_profile["attempts"][0]["selected_profile"]["profile"]["model_id"] = "wrong-model" + changed_status = json.loads(original) + changed_status["attempts"][0]["status"] = "succeeded" + mutations = [ + {**receipt, "attempts": None}, + {**receipt, "attempts": []}, + {**receipt, "state": "succeeded"}, + {**receipt, "run_id": "another-run"}, + changed_profile, changed_status, + ] + # Corrupt only negative evidence; each real failed run and its + # opaque native streams were produced through the public runtime. + for changed in mutations: + with self.subTest(receipt=changed): + content = canonical_json(changed).encode() + path.write_bytes(content) + store.connection.execute( + "UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (digest(changed), len(content), artifact["id"]), + ) + try: + with self.assertRaises(ContractError): + fixture.service.policy_evaluate(fixture.spec) + finally: + path.write_bytes(original) + store.connection.execute( + "UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (artifact["sha256"], artifact["byte_size"], artifact["id"]), + ) + finally: + store.close() + + def test_success_cannot_hide_missing_review_behind_failure_metadata(self): + run_id = self.fixture.run_arm("eval-1", "candidate") + store = self.fixture.store() + try: + attempt = store.connection.execute( + "SELECT id,output_metadata FROM attempts WHERE run_id=?", (run_id,), + ).fetchone() + artifact_name = f"review-attempt-{attempt['id']}.json" + metadata = json.loads(attempt["output_metadata"]) + metadata["failure"] = {"code": "CLI_ERROR"} + store.connection.execute( + "UPDATE artifacts SET name='hidden-review.json' WHERE run_id=? AND name=?", + (run_id, artifact_name), + ) + store.connection.execute( + "UPDATE attempts SET output_metadata=? WHERE id=?", + (canonical_json(metadata), attempt["id"]), + ) + with self.assertRaisesRegex(ContractError, "failed reviewer receipt"): + self.fixture.service.policy_evaluate(self.fixture.spec) + finally: + store.close() + def test_reader_uses_prelaunch_snapshot_not_later_mutable_snapshot(self): self.fixture.run_all() run_id = self.fixture.runs[("hold-1", "candidate")] From 4e6297fc53b8dd5a869767c8a3aa2d24ff96f077 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 14:19:29 -0700 Subject: [PATCH 143/197] WIP checkpoint: Connect current experiment eligibility and append-only reviews with public offline fixtures (2026-10-01 14:19) --- docs/generated/core-reference.md | 2 +- docs/plans/engineering-team/CONTRACTS.md | 30 ++ docs/plans/engineering-team/RESUME.md | 48 ++- docs/plans/engineering-team/backlog.json | 12 +- .../R3b1-failure-path-2026-10-01.json | 7 +- .../R3b2-eligibility-partial-2026-10-01.json | 34 ++ plugin/core/src/devsquad/cli.py | 7 + .../src/devsquad/experiment_eligibility.py | 117 +++++++ plugin/core/src/devsquad/learning.py | 37 +- .../015_experiment_evaluation_revisions.sql | 18 + plugin/core/src/devsquad/service.py | 10 +- plugin/core/src/devsquad/store.py | 319 ++++++++++-------- test/core/experiment_runtime_fixture.py | 37 +- test/core/test_capacity.py | 2 +- test/core/test_cli.py | 3 +- test/core/test_decision_store.py | 2 +- test/core/test_experiment_assignment_store.py | 2 +- test/core/test_experiment_eligibility.py | 120 +++++++ test/core/test_handoff_store.py | 4 +- test/core/test_learning.py | 116 +++---- test/core/test_lifecycle.py | 82 +---- test/core/test_store.py | 8 +- 22 files changed, 702 insertions(+), 315 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R3b2-eligibility-partial-2026-10-01.json create mode 100644 plugin/core/src/devsquad/experiment_eligibility.py create mode 100644 plugin/core/src/devsquad/migrations/015_experiment_evaluation_revisions.sql create mode 100644 test/core/test_experiment_eligibility.py diff --git a/docs/generated/core-reference.md b/docs/generated/core-reference.md index 617266f..3fc28bc 100644 --- a/docs/generated/core-reference.md +++ b/docs/generated/core-reference.md @@ -23,7 +23,7 @@ Regenerate with `python3 scripts/generate-core-reference.py`; verify with - `squad outcome [-h] {add} ...` - `squad outcome add [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR] run` - `squad policy [-h] {evaluate} ...` -- `squad policy evaluate [-h] --experiment EXPERIMENT [--json] [--runtime-dir RUNTIME_DIR]` +- `squad policy evaluate [-h] --experiment EXPERIMENT [--revision-id REVISION_ID] [--previous-evaluation-sha256 PREVIOUS_EVALUATION_SHA256] [--json] [--runtime-dir RUNTIME_DIR]` - `squad prepare [-h] [--cwd CWD] [--model MODEL] [--effort EFFORT] [--permission {read_only,workspace_write}] [--timeout TIMEOUT] [--transport {cli_exec,native_protocol}] [--catalog-file CATALOG_FILE] --prompt PROMPT {codex,antigravity,grok}` - `squad profile [-h] {template-add,binding-bootstrap,qualification-add,binding-change,binding-fallback,binding-show} ...` - `squad profile binding-bootstrap [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index a7f9467..11290d7 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -296,3 +296,33 @@ promotion, rollback or fallback must be checked separately from historical read/replay. Implementation progress and remaining reader/eligibility/public controller work are tracked in [RESUME.md](RESUME.md), not implied by this contract or by pure normalized-chain unit fixtures. + +### Current eligibility and explicit evaluation revisions + +Historical evidence and present authority are distinct. One transaction-bound +eligibility result identifies the experiment, spec SHA256, pinned evaluation +SHA256, saved/current evidence SHA256 and concrete ineligibility reasons. +Legacy v1 evaluations remain readable but cannot authorize a new qualification, +promotion, regression rollback or qualification-backed catalog fallback. +Changed final/correction/attempt evidence invalidates a prior evaluation; it +does not rewrite it or automatically change a binding. Exact completed binding +decision replay returns its original receipt without another mutation. + +R3b.2's agreed implementation contract is an explicit append-only review with +`policy evaluate --experiment FILE --revision-id ID --previous-evaluation-sha256 SHA`. +Both revision arguments are required together. The predecessor must be the +latest saved evaluation, checked in the same write transaction. A revision +uses the original frozen spec and assigned runs, never relabelled new samples. +Schema 15 adds revision history beside the unchanged original `experiments` +row. A repeated revision ID is idempotent only for the same experiment/spec +and predecessor, and still checks present evidence. Responses identify the +revision and predecessor; qualification and decision records pin the existing +evaluation SHA256, which identifies either the original or reviewed revision. + +New qualifications must match the full tested candidate fingerprint and all +assigned candidate tasks' declared role/task class. Qualification replay and +every new binding mutation recheck current evidence within their write +transaction. A proven bootstrap predecessor without qualification retains its +existing explicit baseline contract. Installation of this schema remains +gated on old active/recoverable-run upgrade safety; adding the migration does +not establish that installation gate. Implementation status is in RESUME.md. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 2b04b79..2fe866f 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -4,6 +4,39 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ## Current position — October 1, 2026 +### Latest runtime-repair slice — shared eligibility and explicit review + +R3b.1's unchanged-source gate at `a4a87fd` passed 427 tests (two optional-SDK +skips). The subsequent **partial R3b.2/R3c** slice now adds schema 15's +append-only evaluation revisions and one transaction-bound current-evidence +gate for evaluation replay, qualification/replay, new promotion, regression +rollback and qualification-backed catalog fallback. Exact completed binding +decision replay remains historical and does not mutate again. Qualifications +match the full tested candidate and frozen role/task-class context. Proposal +inputs carry current eligibility; legacy v1 evidence cannot produce authority. + +The two-test red baseline reproduced stale qualification (`ContractError not +raised`) and the missing explicit revision operation. Six new eligibility +tests passed in 32.393 seconds. Lifecycle positive fixtures now use real +offline workers and public host completion rather than SQL-terminalized empty +runs: 14 lifecycle/assignment tests passed in 22.413 seconds. Learning's +positive tests retain their original two-evaluation/one-held-out gate and now +use six distinct actual runs. The latest 23-test learning/eligibility/reader +gate passed in 91.779 seconds; 227 Bash assertions and generated reference +passed. No full integration gate or independent audit has run on schema 15. +The source checkpoints, failure history and limitations are recorded in +[R3b.2 partial evidence](evidence/R3b2-eligibility-partial-2026-10-01.json). + +**Exact next action:** finish the negative/public proof for stale catalog +fallback and regression rollback, correction/qualification transaction races, +revision-chain/CLI semantics and upgraded historical duplicate-outcome report/ +proposal readability. Audit the shared gate and finish R3c's remaining public +coverage, then run one full unchanged-source integration gate at the coherent +R3 boundary. Preserve the original evaluations and completed decisions. Do +not install schema 15 before the old active/recoverable-run upgrade test; R4–R6 +and R8's Claude/Grok/Gemini installed proofs remain pending. No provider call, +installation refresh, purchase, global setting change or push occurred. + ### Review correction and next action Interrupted R3b.1 work newer than the planning checkpoint is now preserved as @@ -37,9 +70,13 @@ success-integrity and saved-evidence tests passed in 9.803 seconds. The Bash gate passed all 227 assertions. Source fingerprints and red/green results are saved in [failure-path evidence](evidence/R3b1-failure-path-2026-10-01.json). -**Exact next action:** run the complete core gate on the checkpointed reader -source before accepting R3b.1; no test is running at this checkpoint. Then -finish R3b.2's shared current-evidence eligibility and +The unchanged-source core gate at `a4a87fd` completed **427 tests in 260.966 +seconds, OK with two optional-SDK skips**. UTC and monotonic wrapper elapsed +times agreed (261.147 seconds). The known SQLite finalizer warning appeared +and remains R5; the gate did not establish its repair. R3b.1 is source/offline +verified, with the independent reader audit still outstanding. + +**Exact next action:** implement R3b.2's shared current-evidence eligibility and explicit append-only evaluation/review revisions; legacy/public compatibility remains R3c and the production paired-trial controller remains R5. @@ -467,9 +504,8 @@ provider paths must not be advertised as verified. 1. Check Git status and recent commits, preserving work newer than this note. Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. Finish R3b.1 from the preserved reader: honest terminal failed-output - handling is repaired; run the full unchanged-source gate. - Then implement R3b.2 shared current-evidence eligibility and append-only +2. R3b.1's source/offline gate passed at `a4a87fd`; do not repeat unchanged + tests. Implement R3b.2 shared current-evidence eligibility and append-only evaluation/review revisions. Validate every relevant attempt, including fallbacks/repairs. Finish R3b/R3c saved evidence, replay, diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 207f019..3781143 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -14,15 +14,15 @@ "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", "next_milestone": "M5", - "next_work_package": "R3b.1", + "next_work_package": "R3b.2", "partial_implementation_checkpoint": { "recorded_on": "2026-10-01", "baseline_revision": "a22b949", "status": "partial", - "artifact": "evidence/R3b1-failure-path-2026-10-01.json", - "scope": "Authoritative v2 saved-run reader, replay invalidation, prelaunch controlled-input guard and public offline fixture; no install or provider calls", - "next_action": "Terminal-failure decoding reproduced and repaired; run full core integration, then R3b.2 shared current-evidence eligibility and append-only revisions", - "limitations": "Full integration and independent reader audit not complete; production paired-trial controller remains R5; native Gemini worker and refreshed installed proof remain unverified" + "artifact": "evidence/R3b2-eligibility-partial-2026-10-01.json", + "scope": "Reader full gate passed at a4a87fd; schema-15 explicit revisions and shared lifecycle eligibility implemented with focused public fixtures; no install or provider calls", + "next_action": "Finish stale fallback/rollback, transaction race, revision/CLI and upgraded legacy public proof; audit R3 and run its complete integration gate", + "limitations": "Schema-15 full integration and independent audit incomplete; SQLite warning and production paired-trial controller remain R5; native Gemini worker and refreshed installed proof unverified" }, "planning_checkpoint": { "recorded_on": "2026-10-01", @@ -44,7 +44,7 @@ "review_work_packages": [ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, - {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3b1-failure-path-2026-10-01.json", "checkpoint": "R3a remains verified at 672e383 with its complete 402-test gate. R3b.1 has a public v2 saved-run reader, changed-evidence replay rejection and controlled-input launch guard. Imported-profile mismatch and honest terminal-failure decoding were reproduced and repaired. New failure-receipt and missing-success regressions pass; full integration and independent reader audit remain incomplete. Shared eligibility and append-only revisions remain R3b.2; legacy/public compatibility remains R3c. R3 is not closed."}, + {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3b2-eligibility-partial-2026-10-01.json", "checkpoint": "R3a verified at 672e383; R3b.1 reader and failed-output repair passed 427-test full gate at a4a87fd. R3b.2 schema-15 explicit revisions and shared lifecycle eligibility have six new passing tests. Real public offline lifecycle/learning positives replace SQL-terminalized runs without lowering pair gates. Latest 23-test focused gate passes. Stale fallback/rollback, race, revision/CLI and upgraded legacy public proof plus full integration/independent audit remain open. R3 is not closed."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, {"id": "R6", "title": "Normal terminal experience and readiness", "status": "pending", "milestones": ["M5", "M7"], "depends_on": ["R1", "R2", "R4", "R5"], "items": ["G3", "G4"]}, diff --git a/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json index 40c88fc..3e1602b 100644 --- a/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R3b1-failure-path-2026-10-01.json @@ -1,7 +1,7 @@ { "schema_version": 1, "work_package": "R3b.1", - "status": "failure_path_repaired_full_gate_pending", + "status": "source_offline_verified_independent_audit_pending", "recorded_at": "2026-10-01T20:57:36Z", "baseline_revision": "bd0cf5c", "source_sha256": { @@ -19,10 +19,11 @@ "integrity_result": "6 tests passed in 9.803 seconds, exit 0", "bash": "bash test/run.sh: 227 assertions in 11 files passed", "generated_reference": "python3 scripts/generate-core-reference.py --check passed", - "full_core": "Pending; run on this checkpointed source", + "full_core": "a4a87fd unchanged source: 427 tests in 260.966 seconds, OK with 2 optional-SDK skips, exit 0; UTC elapsed 261.146617 seconds and monotonic 261.146937 seconds", + "full_core_warning": "Known unclosed SQLite finalizer ResourceWarning remains R5; not suppressed or counted as repaired", "independent_review": "Not completed; no independent audit claimed" }, "fixture_boundary": "Real offline workers and public terminal completion; test-only predeclared-assignment seam, manual outcome addition and synthetic host dispositions. Not the R5 production trial controller or native quality proof.", "external_actions": "No provider request, installation refresh, purchase, reset, global setting change or push", - "next_action": "Full core regression gate, then R3b.2 shared current-evidence eligibility and explicit append-only evaluation/review revisions" + "next_action": "R3b.2 shared current-evidence eligibility and explicit append-only evaluation/review revisions; bounded independent reader audit remains outstanding" } diff --git a/docs/plans/engineering-team/evidence/R3b2-eligibility-partial-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b2-eligibility-partial-2026-10-01.json new file mode 100644 index 0000000..bfc69b0 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3b2-eligibility-partial-2026-10-01.json @@ -0,0 +1,34 @@ +{ + "schema_version": 1, + "work_package": "R3b.2/R3c", + "status": "partial", + "recorded_at": "2026-10-01T21:18:00Z", + "baseline_revision": "a4a87fd", + "source_sha256": { + "plugin/core/src/devsquad/experiment_eligibility.py": "3f53719032bf0fbeb160b1c1c8a82378c9c7a66d8e20b95b76fb292c8bd1691d", + "plugin/core/src/devsquad/store.py": "a7ec8170ca06a5275ad122d27c74d566b18bf8376e4abb30905f531e33ac9151", + "plugin/core/src/devsquad/learning.py": "358df4f1cb4751c9cbb739439d432f82f9b9dc88389975b2fb1ebbf14771850e", + "plugin/core/src/devsquad/migrations/015_experiment_evaluation_revisions.sql": "b5dc3cb747d4fed29dbac615e1b84c869417bdda51d88153efd5e72677b109b4", + "test/core/test_experiment_eligibility.py": "59663e05da53c52d2de483bf8663ab25f4c9086efd50b5bee196946dd357570f" + }, + "verification": { + "red": "2 tests in 10.631 seconds: stale qualification failed to raise ContractError, explicit revision errored because Service did not accept revision_id; exit 1", + "eligibility": "6 tests in 32.393 seconds passed: stale qualification/replay/promotion, exact completed decision replay, profile fingerprint/task context and append-only correction review", + "lifecycle_assignment": "14 tests in 22.413 seconds passed, including real-worker qualified promotion/regression rollback and historical schema-13 byte preservation", + "latest_command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_learning test_experiment_eligibility test_experiment_saved_runs -v", + "latest_result": "23 tests in 91.779 seconds passed", + "bash": "227 assertions in 11 files passed", + "generated_reference": "Regenerated and --check passed", + "full_core": "Not run on schema-15 source; a4a87fd's prior 427-test gate is not substituted", + "independent_audit": "Not completed" + }, + "fixture_boundary": "Positive lifecycle and learning authority now uses real offline worker/check subprocesses and public host completion. Original pair gates retained (lifecycle 1+1; learning 2+1). A test-only assignment seam and manual outcome import remain until R5; no native quality or public production trial-controller claim.", + "remaining": [ + "Stale catalog-fallback and regression-rollback negative/public tests", + "Correction-versus-qualification transaction race proof", + "Revision chain, explicit CLI and upgraded historical duplicate-outcome audit proof", + "R3c remaining workflow coverage, full integration and bounded independent audit", + "R4-R6 runtime connections, old active/recoverable-run safe schema upgrade and R8 installed/live gates" + ], + "external_actions": "No provider call, API request, installation, purchase, reset, global setting change or push" +} diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 3686e92..a8c7c9a 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -304,6 +304,11 @@ def command_report(args: argparse.Namespace) -> tuple[dict, int]: def command_policy_evaluate(args: argparse.Namespace) -> tuple[dict, int]: experiment = _read_json(args.experiment, "experiment file") + if args.revision_id is not None or args.previous_evaluation_sha256 is not None: + return envelope(data=_service(args).policy_evaluate( + experiment, revision_id=args.revision_id, + previous_evaluation_sha256=args.previous_evaluation_sha256, + )), 0 return envelope(data=_service(args).policy_evaluate(experiment)), 0 @@ -503,6 +508,8 @@ def parser() -> argparse.ArgumentParser: policy_sub = policy.add_subparsers(dest="policy_command", required=True) evaluate = policy_sub.add_parser("evaluate") evaluate.add_argument("--experiment", required=True) + evaluate.add_argument("--revision-id") + evaluate.add_argument("--previous-evaluation-sha256") evaluate.add_argument("--json", action="store_true") evaluate.add_argument("--runtime-dir", default=runtime_default) evaluate.set_defaults(func=command_policy_evaluate) diff --git a/plugin/core/src/devsquad/experiment_eligibility.py b/plugin/core/src/devsquad/experiment_eligibility.py new file mode 100644 index 0000000..6e81ef4 --- /dev/null +++ b/plugin/core/src/devsquad/experiment_eligibility.py @@ -0,0 +1,117 @@ +"""One transaction-bound current-evidence gate for all lifecycle consumers.""" + +from __future__ import annotations + +from datetime import datetime +import hashlib +import sqlite3 +from typing import Any + +from .claude_identity import strict_json +from .contracts import ContractError +from .experiment_evidence import read_experiment_chains +from .learning import evaluate_experiment, validate_experiment +from .store import canonical_json + + +def saved_evaluation( + connection: sqlite3.Connection, experiment_id: str, + evaluation_sha256: str | None = None, +) -> dict[str, Any] | None: + """Decode historical bytes without giving them new execution authority.""" + original = connection.execute( + "SELECT * FROM experiments WHERE experiment_id=?", (experiment_id,), + ).fetchone() + if original is None: + return None + record = {**dict(original), "revision_id": None, "previous_evaluation_sha256": None} + if evaluation_sha256 != original["evaluation_sha256"]: + if evaluation_sha256 is None: + revision = connection.execute( + "SELECT * FROM experiment_evaluation_revisions WHERE experiment_id=? ORDER BY id DESC LIMIT 1", + (experiment_id,), + ).fetchone() + else: + revision = connection.execute( + "SELECT * FROM experiment_evaluation_revisions WHERE experiment_id=? AND evaluation_sha256=?", + (experiment_id, evaluation_sha256), + ).fetchone() + if revision is not None: + record.update({key: revision[key] for key in ( + "revision_id", "previous_evaluation_sha256", "evaluation_json", + "evaluation_sha256", "verdict", "recorded_at", + )}) + elif evaluation_sha256 is not None: + return None + for field in ("spec", "evaluation"): + raw = record[f"{field}_json"] + if hashlib.sha256(raw.encode()).hexdigest() != record[f"{field}_sha256"]: + raise ContractError(f"saved experiment {field} hash is invalid") + value = strict_json(raw) + if not isinstance(value, dict): + raise ContractError(f"saved experiment {field} is not an object") + record[field] = value + evaluation = record["evaluation"] + if (evaluation.get("experiment_id") != experiment_id + or evaluation.get("spec_sha256") != record["spec_sha256"] + or evaluation.get("evaluated_at") != record["recorded_at"] + or evaluation.get("verdict") != record["verdict"]): + raise ContractError("saved experiment evaluation identity is invalid") + return record + + +def current_evidence( + connection: sqlite3.Connection, record: dict[str, Any], *, now: datetime, +) -> dict[str, Any]: + """Recompute authority in the caller's transaction, never from a label.""" + if not connection.in_transaction: + raise ContractError("current experiment eligibility requires a ledger transaction") + evaluation = record["evaluation"] + result = { + "schema_version": 1, "experiment_id": record["experiment_id"], + "spec_sha256": record["spec_sha256"], + "evaluation_sha256": record["evaluation_sha256"], + "saved_evidence_sha256": evaluation.get("evidence_sha256"), + "current_evidence_sha256": None, "eligible": False, "reasons": [], + } + if record["spec"].get("schema_version") != 2: + result["reasons"] = ["legacy_unverified_evidence"] + return result + project = connection.execute( + "SELECT git_common_dir FROM projects WHERE id=?", (record["project_id"],), + ).fetchone() + if project is None: + result["reasons"] = ["saved_project_unavailable"] + return result + try: + spec = validate_experiment(record["spec"]) + chains = read_experiment_chains( + connection, spec=spec, project_id=record["project_id"], + project_common_dir=project["git_common_dir"], evaluated_at=now.isoformat(), + ) + fresh = evaluate_experiment( + spec, chains, evaluated_at=now.isoformat(), + project_common_dir=project["git_common_dir"], + ) + except ContractError: + result["reasons"] = ["invalid_saved_run_evidence"] + return result + result["current_evidence_sha256"] = fresh["evidence_sha256"] + if canonical_json({**fresh, "evaluated_at": record["recorded_at"]}) != canonical_json(evaluation): + result["reasons"] = ["saved_evaluation_stale_evidence_changed"] + return result + result["eligible"] = True + return result + + +def require_current_evidence( + connection: sqlite3.Connection, experiment_id: str, evaluation_sha256: str, + *, now: datetime, +) -> tuple[dict[str, Any], dict[str, Any]]: + record = saved_evaluation(connection, experiment_id, evaluation_sha256) + if record is None: + raise ContractError("experiment evaluation evidence is unavailable") + eligibility = current_evidence(connection, record, now=now) + if not eligibility["eligible"]: + raise ContractError("experiment evidence is not current: " + ", ".join(eligibility["reasons"])) + return record, eligibility diff --git a/plugin/core/src/devsquad/learning.py b/plugin/core/src/devsquad/learning.py index 4d34f0d..d50e888 100644 --- a/plugin/core/src/devsquad/learning.py +++ b/plugin/core/src/devsquad/learning.py @@ -685,9 +685,19 @@ def build_learning_proposal( experiment_evidence = None if experiment_record is not None: if (not isinstance(experiment_record, dict) - or set(experiment_record) != EXPERIMENT_RECORD_FIELDS): + or set(experiment_record) not in ( + EXPERIMENT_RECORD_FIELDS, EXPERIMENT_RECORD_FIELDS | {"eligibility"}, + )): raise ContractError("learning proposal experiment record is invalid") - spec = validate_experiment(experiment_record["experiment"]) + raw_spec = experiment_record["experiment"] + if isinstance(raw_spec, dict) and raw_spec.get("schema_version") == 1: + # Audit history as saved, including cases whose old labels reused + # outcomes. Strict new-spec validation is not historical decoding. + if set(raw_spec) != EXPERIMENT_FIELDS: + raise ContractError("historical experiment fields are invalid") + spec = json.loads(canonical_json(raw_spec)) + else: + spec = validate_experiment(raw_spec) evaluation = experiment_record["evaluation"] expected_fields = EXPERIMENT_EVALUATION_FIELDS | ( {"evidence_sha256"} if spec["schema_version"] == 2 else set() @@ -735,6 +745,28 @@ def build_learning_proposal( failures = list(evaluation["failures"]) reasons = list(evaluation["reasons"]) verdict = evaluation["verdict"] + eligibility = experiment_record.get("eligibility") + if spec["schema_version"] == 1: + verdict = "no_change" + reasons = sorted(set([*reasons, "legacy_unverified_evidence"])) + elif eligibility is None: + verdict = "no_change" + reasons = sorted(set([*reasons, "current_evidence_not_checked"])) + else: + if (not isinstance(eligibility, dict) + or eligibility.get("experiment_id") != spec["experiment_id"] + or eligibility.get("spec_sha256") != spec_sha256 + or eligibility.get("evaluation_sha256") != evaluation_sha256 + or eligibility.get("saved_evidence_sha256") != evaluation["evidence_sha256"] + or type(eligibility.get("eligible")) is not bool + or not isinstance(eligibility.get("reasons"), list) + or (eligibility["eligible"] and ( + eligibility.get("current_evidence_sha256") != evaluation["evidence_sha256"] + or eligibility["reasons"]))): + raise ContractError("learning proposal current eligibility is invalid") + if not eligibility["eligible"]: + verdict = "no_change" + reasons = sorted(set([*reasons, *eligibility["reasons"]])) evidence_availability = spec["evidence_availability"] experiment_samples = { split: { @@ -748,6 +780,7 @@ def build_learning_proposal( "spec_sha256": spec_sha256, "evaluation_sha256": evaluation_sha256, "recorded_at": experiment_record["recorded_at"], + "eligibility": eligibility, } identity = { diff --git a/plugin/core/src/devsquad/migrations/015_experiment_evaluation_revisions.sql b/plugin/core/src/devsquad/migrations/015_experiment_evaluation_revisions.sql new file mode 100644 index 0000000..d0785ea --- /dev/null +++ b/plugin/core/src/devsquad/migrations/015_experiment_evaluation_revisions.sql @@ -0,0 +1,18 @@ +-- Never overwrite the original evaluation. Explicit reviews pin a predecessor +-- and reuse the original frozen spec/assignments, not new independent samples. +CREATE TABLE experiment_evaluation_revisions ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + revision_id TEXT NOT NULL UNIQUE, + experiment_id TEXT NOT NULL REFERENCES experiments(experiment_id), + previous_evaluation_sha256 TEXT NOT NULL + CHECK(length(previous_evaluation_sha256) = 64 AND previous_evaluation_sha256 NOT GLOB '*[^0-9a-f]*'), + evaluation_json TEXT NOT NULL, + evaluation_sha256 TEXT NOT NULL + CHECK(length(evaluation_sha256) = 64 AND evaluation_sha256 NOT GLOB '*[^0-9a-f]*'), + verdict TEXT NOT NULL CHECK(verdict IN ('no_change', 'promotion_proposal')), + recorded_at TEXT NOT NULL, + UNIQUE(experiment_id, evaluation_sha256) +); + +CREATE INDEX experiment_evaluation_review_history +ON experiment_evaluation_revisions(experiment_id, id); diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index b2f21d0..616e68c 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -999,11 +999,17 @@ def learning_report(self, project: str | Path) -> dict[str, Any]: finally: store.close() - def policy_evaluate(self, experiment: dict[str, Any]) -> dict[str, Any]: + def policy_evaluate( + self, experiment: dict[str, Any], *, revision_id: str | None = None, + previous_evaluation_sha256: str | None = None, + ) -> dict[str, Any]: """Evaluate and save one frozen learning experiment without promotion.""" store = self._store() try: - return store.evaluate_learning_experiment(experiment) + return store.evaluate_learning_experiment( + experiment, revision_id=revision_id, + previous_evaluation_sha256=previous_evaluation_sha256, + ) finally: store.close() diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 27e5480..d4de731 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -17,7 +17,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 14 +SUPPORTED_SCHEMA_VERSION = 15 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -1762,10 +1762,20 @@ def learning_report( def evaluate_learning_experiment( self, experiment: dict[str, Any], *, now: datetime | None = None, + revision_id: str | None = None, + previous_evaluation_sha256: str | None = None, ) -> dict[str, Any]: - """Persist one deterministic, replay-safe experiment evaluation.""" + """Persist an original evaluation or an explicit append-only review.""" + from .experiment_eligibility import current_evidence, saved_evaluation + from .experiment_provenance import require_sha256 from .learning import evaluate_experiment, validate_experiment + if (revision_id is None) != (previous_evaluation_sha256 is None): + raise ContractError("evaluation revision id and predecessor must be supplied together") + if revision_id is not None: + if not isinstance(revision_id, str) or not revision_id.strip() or len(revision_id) > 128: + raise ContractError("evaluation revision id is invalid") + require_sha256(previous_evaluation_sha256, "previous evaluation") spec = validate_experiment(experiment) project_path = Path(spec["project_path"]).resolve(strict=True) # V2 hashes the exact predeclared specification. Resolving a symlink @@ -1778,22 +1788,40 @@ def evaluate_learning_experiment( common_dir = git_common_dir(project_path) self.connection.execute("BEGIN IMMEDIATE") try: - existing = self.connection.execute( - "SELECT spec_json,evaluation_json,evaluation_sha256,recorded_at " - "FROM experiments WHERE experiment_id=?", - (spec["experiment_id"],), - ).fetchone() + existing = saved_evaluation(self.connection, spec["experiment_id"]) if existing is not None and existing["spec_json"] != spec_json: raise ConflictError( "experiment id was already used with a different specification", ) - if existing is not None and spec["schema_version"] == 1: + if revision_id is not None: + if existing is None or spec["schema_version"] != 2: + raise ContractError("explicit evaluation review requires an original v2 evaluation") + revision = self.connection.execute( + "SELECT experiment_id,previous_evaluation_sha256,evaluation_sha256 " + "FROM experiment_evaluation_revisions WHERE revision_id=?", (revision_id,), + ).fetchone() + if revision is not None: + if (revision["experiment_id"] != spec["experiment_id"] + or revision["previous_evaluation_sha256"] != previous_evaluation_sha256): + raise ConflictError("evaluation revision id was already used with different evidence") + existing = saved_evaluation(self.connection, spec["experiment_id"], revision["evaluation_sha256"]) + else: + if existing["evaluation_sha256"] != previous_evaluation_sha256: + raise ConflictError("evaluation revision predecessor is no longer the latest evaluation") + existing = None + if existing is not None: + eligibility = current_evidence(self.connection, existing, now=current) + if spec["schema_version"] == 2 and not eligibility["eligible"]: + raise ContractError("experiment evidence changed; saved evaluation is stale and requires explicit review/revision: " + ", ".join(eligibility["reasons"])) self.connection.execute("COMMIT") return { "experiment": spec, - "evaluation": json.loads(existing["evaluation_json"]), + "evaluation": existing["evaluation"], "evaluation_sha256": existing["evaluation_sha256"], "recorded_at": existing["recorded_at"], + "revision_id": existing["revision_id"], + "previous_evaluation_sha256": existing["previous_evaluation_sha256"], + "eligibility": eligibility, "replayed": True, } project = self.connection.execute( @@ -1836,47 +1864,39 @@ def evaluate_learning_experiment( spec, chains, evaluated_at=current.isoformat(), project_common_dir=str(common_dir), ) - if existing is not None: - try: - saved = json.loads(existing["evaluation_json"]) - except (ValueError, TypeError) as exc: - raise ContractError("saved experiment evaluation is invalid") from exc - if (not isinstance(saved, dict) - or saved.get("evaluated_at") != existing["recorded_at"] - or hashlib.sha256(existing["evaluation_json"].encode()).hexdigest() - != existing["evaluation_sha256"]): - raise ContractError("saved experiment evaluation hash/identity is invalid") - # Receipts are immutable. Corrections or changed run evidence - # require an explicit revision, never a silently updated replay. - if canonical_json({**evaluation, "evaluated_at": saved["evaluated_at"]}) != canonical_json(saved): - raise ContractError( - "experiment evidence changed; saved evaluation is stale and requires explicit review/revision", - ) - self.connection.execute("COMMIT") - return { - "experiment": spec, "evaluation": saved, - "evaluation_sha256": existing["evaluation_sha256"], - "recorded_at": existing["recorded_at"], "replayed": True, - } evaluation_json = canonical_json(evaluation) evaluation_sha256 = hashlib.sha256(evaluation_json.encode()).hexdigest() recorded_at = current.isoformat() - self.connection.execute( - "INSERT INTO experiments(experiment_id,project_id,project_path,spec_json," - "spec_sha256,evaluation_json,evaluation_sha256,verdict,recorded_at) " - "VALUES(?,?,?,?,?,?,?,?,?)", - ( - spec["experiment_id"], project_id, spec["project_path"], - spec_json, spec_sha256, evaluation_json, evaluation_sha256, - evaluation["verdict"], recorded_at, - ), - ) + if revision_id is None: + self.connection.execute( + "INSERT INTO experiments(experiment_id,project_id,project_path,spec_json," + "spec_sha256,evaluation_json,evaluation_sha256,verdict,recorded_at) " + "VALUES(?,?,?,?,?,?,?,?,?)", + ( + spec["experiment_id"], project_id, spec["project_path"], + spec_json, spec_sha256, evaluation_json, evaluation_sha256, + evaluation["verdict"], recorded_at, + ), + ) + else: + self.connection.execute( + "INSERT INTO experiment_evaluation_revisions(revision_id,experiment_id," + "previous_evaluation_sha256,evaluation_json,evaluation_sha256,verdict,recorded_at) " + "VALUES(?,?,?,?,?,?,?)", + (revision_id, spec["experiment_id"], previous_evaluation_sha256, + evaluation_json, evaluation_sha256, evaluation["verdict"], recorded_at), + ) + saved = saved_evaluation(self.connection, spec["experiment_id"], evaluation_sha256) + eligibility = current_evidence(self.connection, saved, now=current) self.connection.execute("COMMIT") return { "experiment": spec, "evaluation": evaluation, "evaluation_sha256": evaluation_sha256, "recorded_at": recorded_at, + "revision_id": revision_id, + "previous_evaluation_sha256": previous_evaluation_sha256, + "eligibility": eligibility, "replayed": False, } except Exception: @@ -1887,6 +1907,7 @@ def learning_proposal_inputs( self, project: Path, *, now: datetime | None = None, ) -> dict[str, Any]: """Read a consistent report and the latest frozen project experiment.""" + from .experiment_eligibility import current_evidence, saved_evaluation project_path = project.resolve(strict=True) current = _authoritative_now(now) self.connection.execute("BEGIN") @@ -1894,26 +1915,26 @@ def learning_proposal_inputs( report = self.learning_report(project_path, now=current) if report["project_id"] is None: row = self.connection.execute( - "SELECT spec_json,spec_sha256,evaluation_json,evaluation_sha256," - "recorded_at FROM experiments WHERE project_path=? " + "SELECT experiment_id FROM experiments WHERE project_path=? " "ORDER BY recorded_at DESC,id DESC LIMIT 1", (str(project_path),), ).fetchone() else: row = self.connection.execute( - "SELECT spec_json,spec_sha256,evaluation_json,evaluation_sha256," - "recorded_at FROM experiments WHERE project_id=? OR " + "SELECT experiment_id FROM experiments WHERE project_id=? OR " "(project_id IS NULL AND project_path=?) " "ORDER BY recorded_at DESC,id DESC LIMIT 1", (report["project_id"], str(project_path)), ).fetchone() - experiment = None if row is None else { - "experiment": json.loads(row["spec_json"]), - "spec_sha256": row["spec_sha256"], - "evaluation": json.loads(row["evaluation_json"]), - "evaluation_sha256": row["evaluation_sha256"], - "recorded_at": row["recorded_at"], - } + experiment = None + if row is not None: + saved = saved_evaluation(self.connection, row["experiment_id"]) + experiment = { + "experiment": saved["spec"], "spec_sha256": saved["spec_sha256"], + "evaluation": saved["evaluation"], "evaluation_sha256": saved["evaluation_sha256"], + "recorded_at": saved["recorded_at"], + "eligibility": current_evidence(self.connection, saved, now=current), + } self.connection.execute("COMMIT") return {"report": report, "experiment": experiment} except Exception: @@ -2088,15 +2109,88 @@ def bootstrap_profile_binding( self.connection.execute("ROLLBACK") raise + def _qualification_evidence( + self, record: dict[str, Any], *, now: datetime, + ) -> tuple[list[str], dict[str, Any] | None]: + """The same current-evidence and context checks for every consumer.""" + from .experiment_eligibility import require_current_evidence + from .lifecycle import qualification_gate_failures, validate_profile_template, profile_fingerprint + + template_row = self.connection.execute( + "SELECT payload_json,payload_sha256 FROM profile_templates WHERE template_id=?", + (record["template_id"],), + ).fetchone() + if template_row is None: + raise ContractError("qualification template is not registered") + if hashlib.sha256(template_row["payload_json"].encode()).hexdigest() != template_row["payload_sha256"]: + raise ContractError("qualification template hash is invalid") + template = validate_profile_template(json.loads(template_row["payload_json"])) + failures = qualification_gate_failures(record, template) + eligibility = None + if record["experiment_id"] is not None: + experiment, eligibility = require_current_evidence( + self.connection, record["experiment_id"], record["evaluation_sha256"], now=now, + ) + spec, evaluation = experiment["spec"], experiment["evaluation"] + variable = spec["variable"] + if (variable["alias"] != record["alias"] + or variable["candidate_profile_id"] != record["candidate_profile"]["id"] + or variable["candidate_profile_sha256"] != profile_fingerprint(record["candidate_profile"])): + raise ContractError("qualification candidate fingerprint does not match the tested experiment") + contexts = self.connection.execute( + "SELECT assignment_json,snapshot_json FROM experiment_assignments " + "WHERE experiment_id=? AND arm='candidate'", (record["experiment_id"],), + ).fetchall() + if not contexts or any( + json.loads(row["assignment_json"])["role"] != variable["role"] + or json.loads(row["snapshot_json"])["task"]["task_class"] != record["task_class"] + for row in contexts): + raise ContractError("qualification role/task-class context does not match actual frozen tasks") + metrics = evaluation["metrics"] + escaped = sum(metrics[split]["candidate_escaped_defects"] for split in ("evaluation", "held_out")) + if (record["measured"]["evaluation_pairs"] != metrics["evaluation"]["available_pairs"] + or record["measured"]["held_out_pairs"] != metrics["held_out"]["available_pairs"] + or record["measured"]["critical_defects"] != escaped): + raise ContractError("qualification measurements do not match saved evaluation") + if evaluation["verdict"] != "promotion_proposal": + failures.append("experiment_did_not_propose_promotion") + failures = sorted(set(failures)) + if record["verdict"] == "qualified" and failures: + raise ContractError("qualification gate did not pass: " + ", ".join(failures)) + return failures, eligibility + + def _require_current_qualification( + self, qualification_id: str, *, now: datetime, + ) -> dict[str, Any]: + from .lifecycle import validate_qualification + + row = self.connection.execute( + "SELECT * FROM qualification_runs WHERE qualification_id=?", (qualification_id,), + ).fetchone() + if row is None or row["verdict"] != "qualified": + raise ContractError("binding change requires a qualified candidate") + if hashlib.sha256(row["payload_json"].encode()).hexdigest() != row["payload_sha256"]: + raise ContractError("saved qualification hash is invalid") + record = validate_qualification(json.loads(row["payload_json"])) + if any(row[key] != record[key] for key in ( + "qualification_id", "alias", "template_id", "experiment_id", "evaluation_sha256", "verdict")) or row["profile_id"] != record["candidate_profile"]["id"]: + raise ContractError("saved qualification identity is invalid") + failures, _ = self._qualification_evidence(record, now=now) + if failures or canonical_json(failures) != row["gate_failures_json"]: + raise ContractError("saved qualification gates are inconsistent") + profile_row = self.connection.execute( + "SELECT profile_json,profile_sha256 FROM concrete_profiles WHERE profile_id=?", (row["profile_id"],), + ).fetchone() + payload, digest = self._profile_record(record["candidate_profile"]) + if profile_row is None or profile_row["profile_json"] != payload or profile_row["profile_sha256"] != digest: + raise ContractError("saved qualification concrete profile differs from tested evidence") + return record + def record_profile_qualification( self, qualification: dict[str, Any], *, now: datetime | None = None, ) -> dict[str, Any]: """Persist bounded qualification evidence after checking saved evaluation.""" - from .lifecycle import ( - qualification_gate_failures, - validate_profile_template, - validate_qualification, - ) + from .lifecycle import validate_qualification record = validate_qualification(qualification) payload = canonical_json(record) @@ -2114,65 +2208,20 @@ def record_profile_qualification( raise ConflictError( "qualification id was already used with different evidence", ) + failures, eligibility = self._qualification_evidence(record, now=_authoritative_now(now)) + if (existing["payload_sha256"] != digest + or canonical_json(failures) != existing["gate_failures_json"]): + raise ContractError("saved qualification evidence hash/gates are invalid") self.connection.execute("COMMIT") return { "qualification": record, "qualification_sha256": existing["payload_sha256"], "gate_failures": json.loads(existing["gate_failures_json"]), "recorded_at": existing["recorded_at"], + "eligibility": eligibility, "replayed": True, } - template_row = self.connection.execute( - "SELECT payload_json FROM profile_templates WHERE template_id=?", - (record["template_id"],), - ).fetchone() - if template_row is None: - raise ContractError("qualification template is not registered") - template = validate_profile_template( - json.loads(template_row["payload_json"]), - ) - failures = qualification_gate_failures(record, template) - experiment = None - if record["experiment_id"] is not None: - experiment = self.connection.execute( - "SELECT spec_json,evaluation_json,evaluation_sha256,verdict " - "FROM experiments WHERE experiment_id=?", - (record["experiment_id"],), - ).fetchone() - if (experiment is None - or experiment["evaluation_sha256"] - != record["evaluation_sha256"]): - raise ContractError( - "qualification experiment evidence is unavailable", - ) - spec = json.loads(experiment["spec_json"]) - evaluation = json.loads(experiment["evaluation_json"]) - if (spec["variable"]["alias"] != record["alias"] - or spec["variable"]["candidate_profile_id"] - != record["candidate_profile"]["id"]): - raise ContractError( - "qualification candidate does not match the experiment", - ) - metrics = evaluation["metrics"] - escaped = sum( - metrics[split]["candidate_escaped_defects"] - for split in ("evaluation", "held_out") - ) - if (record["measured"]["evaluation_pairs"] - != metrics["evaluation"]["available_pairs"] - or record["measured"]["held_out_pairs"] - != metrics["held_out"]["available_pairs"] - or record["measured"]["critical_defects"] != escaped): - raise ContractError( - "qualification measurements do not match saved evaluation", - ) - if experiment["verdict"] != "promotion_proposal": - failures.append("experiment_did_not_propose_promotion") - failures = sorted(set(failures)) - if record["verdict"] == "qualified" and failures: - raise ContractError( - "qualification gate did not pass: " + ", ".join(failures), - ) + failures, eligibility = self._qualification_evidence(record, now=_authoritative_now(now)) self._insert_concrete_profile( record["candidate_profile"], recorded_at, ) @@ -2195,6 +2244,7 @@ def record_profile_qualification( "qualification_sha256": digest, "gate_failures": failures, "recorded_at": recorded_at, + "eligibility": eligibility, "replayed": False, } except Exception: @@ -2305,7 +2355,9 @@ def change_profile_binding( target_profile_sha256 = qualification["profile_sha256"] target_template_sha256 = qualification["template_sha256"] target_qualification_id = request["qualification_id"] - qualification_payload = json.loads(qualification["payload_json"]) + qualification_payload = self._require_current_qualification( + request["qualification_id"], now=_authoritative_now(now), + ) if (request["actor"] == "guarded_auto" and target_template_id != current["template_id"]): raise ContractError( @@ -2340,40 +2392,39 @@ def change_profile_binding( target_template_sha256 = target["template_sha256"] target_qualification_id = target["qualification_id"] if target_qualification_id is not None: - qualified = self.connection.execute( - "SELECT verdict,payload_json FROM qualification_runs " - "WHERE qualification_id=?", - (target_qualification_id,), - ).fetchone() - if qualified is None or qualified["verdict"] != "qualified": - raise ContractError("rollback target is no longer qualified") - qualification_payload = json.loads(qualified["payload_json"]) + qualification_payload = self._require_current_qualification( + target_qualification_id, now=_authoritative_now(now), + ) if (request["actor"] == "guarded_auto" and target_template_id != current["template_id"]): raise ContractError( "guarded automation cannot change lifecycle policy", ) - regression = self.connection.execute( - "SELECT spec_json,evaluation_json,evaluation_sha256,verdict " - "FROM experiments WHERE experiment_id=?", - (request["experiment_id"],), - ).fetchone() - if (regression is None - or regression["evaluation_sha256"] - != request["evaluation_sha256"] - or regression["verdict"] != "no_change"): + from .experiment_eligibility import require_current_evidence + + regression, regression_eligibility = require_current_evidence( + self.connection, request["experiment_id"], request["evaluation_sha256"], + now=_authoritative_now(now), + ) + if regression["verdict"] != "no_change": raise ContractError( "rollback requires saved no-change regression evidence", ) - regression_spec = json.loads(regression["spec_json"]) - regression_evaluation = json.loads(regression["evaluation_json"]) + regression_spec = regression["spec"] + regression_evaluation = regression["evaluation"] variable = regression_spec["variable"] if (variable["alias"] != request["alias"] or variable["candidate_profile_id"] != current["profile_id"] - or variable["control_profile_id"] != target_profile["id"]): + or variable["control_profile_id"] != target_profile["id"] + or variable["candidate_profile_sha256"] != current["profile_sha256"] + or variable["control_profile_sha256"] != target_profile_sha256): raise ContractError( "rollback experiment does not compare the active and target profiles", ) + if any( + regression_evaluation["metrics"][split]["available_pairs"] < regression_spec["gate"][minimum] + for split, minimum in (("evaluation", "min_evaluation_pairs"), ("held_out", "min_held_out_pairs"))): + raise ContractError("rollback requires complete evaluation and held-out regression pairs") rollback_evaluation = { "experiment_id": request["experiment_id"], "evaluation_sha256": request["evaluation_sha256"], @@ -2381,6 +2432,7 @@ def change_profile_binding( "reasons": regression_evaluation["reasons"], "metrics": regression_evaluation["metrics"], "failures": regression_evaluation["failures"], + "eligibility": regression_eligibility, } if target_profile["id"] == current["profile_id"]: raise ConflictError("binding already targets the requested profile") @@ -2567,9 +2619,12 @@ def fallback_unavailable_profile_binding( if (candidate["qualification_verdict"] != "qualified" or json.loads(candidate["qualification_failures"])): continue - candidate_qualification = json.loads( - candidate["qualification_json"], - ) + try: + candidate_qualification = self._require_current_qualification( + candidate["qualification_id"], now=_authoritative_now(now), + ) + except ContractError: + continue else: candidate_qualification = None if (request["actor"] == "guarded_auto" diff --git a/test/core/experiment_runtime_fixture.py b/test/core/experiment_runtime_fixture.py index 8340ffa..e574c84 100644 --- a/test/core/experiment_runtime_fixture.py +++ b/test/core/experiment_runtime_fixture.py @@ -27,13 +27,20 @@ class ExperimentRuntimeFixture: - def __init__(self, root: Path, *, with_fallback=False, fail_candidate=False): + def __init__( + self, root: Path, *, with_fallback=False, fail_candidate=False, + service=None, repo=None, experiment_id="saved-run-review-pair", + candidate_succeeds=True, case_splits=None, + ): self.root = root.resolve() self.root.mkdir(parents=True, exist_ok=True) self.with_fallback = with_fallback self.fail_candidate = fail_candidate - self.repo = self.root / "repo" - self.service = Service(self.root / "runtime") + self.candidate_succeeds = candidate_succeeds + self.experiment_id = experiment_id + self.case_splits = case_splits or [("eval-1", "evaluation"), ("hold-1", "held_out")] + self.repo = repo or self.root / "repo" + self.service = service or Service(self.root / "runtime") self.runs = {} self.git("init", "-q", str(self.repo), outside=True) self.git("config", "user.email", "test@example.invalid") @@ -43,7 +50,7 @@ def __init__(self, root: Path, *, with_fallback=False, fail_candidate=False): self.git("commit", "-qm", "baseline") self.base = self.git("rev-parse", "HEAD").strip() self.targets = {} - for case_id in ("eval-1", "hold-1"): + for case_id, _ in self.case_splits: (self.repo / "README").write_text(f"candidate {case_id}\n") self.git("add", "README") self.git("commit", "-qm", f"candidate {case_id}") @@ -67,19 +74,19 @@ def __init__(self, root: Path, *, with_fallback=False, fail_candidate=False): self.policy["roles"]["reviewer"] = [{"kind": "profile", "id": fallback["id"]}] self.package_path, self.package_digest = self.service._freeze_package() cases = [] - for case_id, split in (("eval-1", "evaluation"), ("hold-1", "held_out")): + for case_id, split in self.case_splits: identities = paired_input_identity( self.declaration_snapshot(case_id, "control"), role="reviewer", package_digest=self.package_digest, ) cases.append({ "case_id": case_id, "split": split, - "control_outcome_id": f"control-{case_id}", - "candidate_outcome_id": f"candidate-{case_id}", + "control_outcome_id": self.outcome_id(case_id, "control"), + "candidate_outcome_id": self.outcome_id(case_id, "candidate"), **identities, }) self.spec = { - "schema_version": 2, "experiment_id": "saved-run-review-pair", + "schema_version": 2, "experiment_id": experiment_id, "project_path": str(self.repo), "question": "Does the candidate fixture improve paired acceptance?", "hypothesis": "Candidate outcomes improve in evaluation and held-out cases.", @@ -96,10 +103,14 @@ def __init__(self, root: Path, *, with_fallback=False, fail_candidate=False): "noninferiority_margin": 0.0, "minimum_success_gain": 1.0, "max_candidate_escaped_defects": 0, }, - "budget": {"max_cases": 2, "max_worker_invocations": 8 if with_fallback else 4, "wall_seconds": 600}, + "budget": {"max_cases": len(cases), "max_worker_invocations": len(cases) * (4 if with_fallback else 2), "wall_seconds": 600}, "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, } + def outcome_id(self, case_id, arm): + suffix = "" if self.experiment_id == "saved-run-review-pair" else f"-{self.experiment_id}" + return f"{arm}-{case_id}{suffix}" + def git(self, *args, outside=False): command = ["git"] if outside else ["git", "-C", str(self.repo)] return subprocess.run(command + list(args), check=True, capture_output=True, text=True).stdout @@ -165,7 +176,7 @@ def predeclared_assignment(*args, **kwargs): if no_attempt: stack.enter_context(patch.object(self.service, "_spawn_daemon", return_value=0)) started = self.service.start( - self.task(case_id, arm), f"{arm}-{case_id}", + self.task(case_id, arm), self.outcome_id(case_id, arm), _internal_review_fixture={"verdict": "clean", "summary": "Fixture review of the frozen candidate.", "findings": []}, ) run_id = started["run_id"] @@ -186,15 +197,15 @@ def predeclared_assignment(*args, **kwargs): packet = claimed["handoff"]["packet"] body = { "schema_version": 1, "submission_id": f"finish-{arm}-{case_id}", - "disposition": "accept" if arm == "candidate" else "reject", + "disposition": "accept" if (arm == "candidate") == self.candidate_succeeds else "reject", "reason": "Predeclared offline fixture disposition.", "evidence_refs": [{"artifact_id": reference["artifact_id"], "sha256": reference["sha256"]} for reference in packet["artifacts"]], } completed = self.service.handoff_complete(run_id, claimed["claim"], {**body, "submission_hash": request_hash(body)}) - verdict = "succeeded" if arm == "candidate" else "failed" + verdict = "succeeded" if (arm == "candidate") == self.candidate_succeeds else "failed" if completed["state"] != verdict: raise AssertionError(f"public fixture disposition failed: {completed}") - outcome = experimental_final(f"{arm}-{case_id}", verdict) + outcome = experimental_final(self.outcome_id(case_id, arm), verdict) outcome["observed_at"] = datetime.now(timezone.utc).isoformat() outcome["evidence_refs"] = ["receipt.json", "result-receipt.json"] self.service.outcome_add(run_id, outcome) diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index a4f421e..10c7f10 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -184,7 +184,7 @@ def test_migration_nine_creates_capacity_ledger(self): version = store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 14) + self.assertEqual(version, 15) tables = { row[0] for row in store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 538cf1b..25d6ba3 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -693,7 +693,8 @@ def test_installed_wheel_contains_and_applies_current_migrations(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 14 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 15 + assert migrations.joinpath("015_experiment_evaluation_revisions.sql").is_file() attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} diff --git a/test/core/test_decision_store.py b/test/core/test_decision_store.py index db9707b..ab17148 100644 --- a/test/core/test_decision_store.py +++ b/test/core/test_decision_store.py @@ -191,7 +191,7 @@ def test_schema_thirteen_contains_decision_cache_and_run_links(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 14) + self.assertEqual(version, 15) tables = { row[0] for row in self.store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_experiment_assignment_store.py b/test/core/test_experiment_assignment_store.py index aba47ea..1e3c6a1 100644 --- a/test/core/test_experiment_assignment_store.py +++ b/test/core/test_experiment_assignment_store.py @@ -189,7 +189,7 @@ def test_schema_13_migration_preserves_legacy_evaluation_bytes(self): saved = upgraded.connection.execute('SELECT * FROM experiments').fetchone() self.assertEqual(saved['spec_json'], spec) self.assertEqual(saved['evaluation_json'], evaluation) - self.assertEqual(upgraded.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0], 14) + self.assertEqual(upgraded.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0], 15) self.assertEqual(upgraded.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 0) diff --git a/test/core/test_experiment_eligibility.py b/test/core/test_experiment_eligibility.py new file mode 100644 index 0000000..1f2d2fc --- /dev/null +++ b/test/core/test_experiment_eligibility.py @@ -0,0 +1,120 @@ +"""New lifecycle authority is bound to current public saved-run evidence.""" + +import copy +from datetime import datetime, timezone +from pathlib import Path +import tempfile +import unittest + +from experiment_runtime_fixture import ExperimentRuntimeFixture +from test_learning import experimental_final +import test_lifecycle as lifecycle_fixtures +from devsquad.contracts import ContractError +from devsquad.store import ConflictError + + +class ExperimentEligibilityTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-eligibility-") + self.addCleanup(self.temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(self.temporary.name)) + self.addCleanup(self.fixture.close) + self.fixture.run_all() + self.evaluation = self.fixture.service.policy_evaluate(self.fixture.spec) + self.store = self.fixture.store() + self.addCleanup(self.store.close) + self.template = lifecycle_fixtures.lifecycle_template() + self.store.bootstrap_profile_binding(self.template, self.fixture.profiles["control"], version=7) + helper = lifecycle_fixtures.ProfileLifecycleTest() + helper.candidate = self.fixture.profiles["candidate"] + self.qualification = helper.qualification(self.evaluation) + self.qualification["budget"].update(max_worker_invocations=4, worker_invocations=4) + + def correct(self): + correction = experimental_final("escaped-held-out", "succeeded") + correction.update( + kind="late_correction", verdict="escaped_defect", + corrects_outcome_id="candidate-hold-1", + observed_at=datetime.now(timezone.utc).isoformat(), + summary="A real later escaped-defect observation invalidates eligibility.", + ) + self.fixture.service.outcome_add(self.fixture.runs[("hold-1", "candidate")], correction) + + def test_correction_after_evaluation_blocks_new_qualification_atomically(self): + self.correct() + with self.assertRaisesRegex(ContractError, "current|stale|changed"): + self.store.record_profile_qualification(self.qualification) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM qualification_runs").fetchone()[0], 0) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 7) + + def test_correction_after_qualification_blocks_replay_and_new_promotion(self): + self.store.record_profile_qualification(self.qualification) + self.correct() + with self.assertRaisesRegex(ContractError, "current|stale|changed"): + self.store.record_profile_qualification(self.qualification) + with self.assertRaisesRegex(ContractError, "current|stale|changed"): + self.store.change_profile_binding(lifecycle_fixtures.ProfileLifecycleTest.promotion("stale-promotion")) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 7) + self.assertEqual(self.store.profile_binding_decisions("review.deep"), []) + + def test_completed_decision_replay_preserves_historical_receipt_after_correction(self): + self.store.record_profile_qualification(self.qualification) + request = lifecycle_fixtures.ProfileLifecycleTest.promotion("completed-promotion") + first = self.store.change_profile_binding(request) + self.correct() + replay = self.store.change_profile_binding(request) + self.assertEqual(replay, {**first, "replayed": True}) + self.assertEqual(self.store.profile_binding("review.deep")["version"], 8) + self.assertEqual(len(self.store.profile_binding_decisions("review.deep")), 1) + + def test_same_profile_id_with_different_fingerprint_cannot_qualify(self): + changed = copy.deepcopy(self.qualification) + changed["candidate_profile"]["required_tools"] = ["read", "web"] + with self.assertRaisesRegex(ContractError, "fingerprint|candidate"): + self.store.record_profile_qualification(changed) + + def test_task_class_claim_must_match_actual_frozen_tasks(self): + template = copy.deepcopy(self.template) + template["template_id"] = "template-multiple-classes" + template["allowed_task_classes"].append("untested-task-class") + self.store.register_profile_template(template) + changed = copy.deepcopy(self.qualification) + changed.update(template_id=template["template_id"], task_class="untested-task-class") + with self.assertRaisesRegex(ContractError, "task.class|context"): + self.store.record_profile_qualification(changed) + + def test_explicit_revision_keeps_original_bytes_and_reuses_original_assignments(self): + original = dict(self.store.connection.execute( + "SELECT * FROM experiments WHERE experiment_id=?", (self.fixture.spec["experiment_id"],), + ).fetchone()) + self.correct() + revision = self.fixture.service.policy_evaluate( + self.fixture.spec, revision_id="review-after-escape", + previous_evaluation_sha256=self.evaluation["evaluation_sha256"], + ) + self.assertEqual(revision["evaluation"]["verdict"], "no_change") + self.assertNotEqual(revision["evaluation_sha256"], self.evaluation["evaluation_sha256"]) + self.assertEqual(revision["revision_id"], "review-after-escape") + self.assertEqual(revision["previous_evaluation_sha256"], self.evaluation["evaluation_sha256"]) + self.assertTrue(revision["eligibility"]["eligible"]) + replay = self.fixture.service.policy_evaluate( + self.fixture.spec, revision_id="review-after-escape", + previous_evaluation_sha256=self.evaluation["evaluation_sha256"], + ) + self.assertEqual(replay, {**revision, "replayed": True}) + saved = dict(self.store.connection.execute( + "SELECT * FROM experiments WHERE experiment_id=?", (self.fixture.spec["experiment_id"],), + ).fetchone()) + self.assertEqual(saved, original) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiment_assignments").fetchone()[0], 4) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiment_evaluation_revisions").fetchone()[0], 1) + with self.assertRaises(ConflictError): + self.fixture.service.policy_evaluate( + self.fixture.spec, revision_id="stale-predecessor-review", + previous_evaluation_sha256=self.evaluation["evaluation_sha256"], + ) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiment_evaluation_revisions").fetchone()[0], 1) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index bef5fcb..3a42fba 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 14, + 15, ) tables = { row[0] @@ -684,7 +684,7 @@ def test_installed_wheel_applies_schema_four_to_twelve(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 14 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 15 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_learning.py b/test/core/test_learning.py index c3573dd..aa480e7 100644 --- a/test/core/test_learning.py +++ b/test/core/test_learning.py @@ -123,6 +123,22 @@ def experimental_final(outcome_id, verdict): class LearningContractTest(unittest.TestCase): + def runtime_fixture(self): + from experiment_runtime_fixture import ExperimentRuntimeFixture + + temporary = tempfile.TemporaryDirectory(prefix="devsquad-learning-public-") + self.addCleanup(temporary.cleanup) + fixture = ExperimentRuntimeFixture( + Path(temporary.name), case_splits=[ + ("eval-1", "evaluation"), ("eval-2", "evaluation"), ("hold-1", "held_out"), + ], + ) + self.addCleanup(fixture.close) + # Retain the original two-evaluation/one-held-out acceptance threshold. + fixture.spec["gate"].update(min_evaluation_pairs=2, minimum_success_gain=0.5) + fixture.run_all() + return fixture + def test_outcome_contract_rejects_false_success_and_mutation(self): normalized = validate_outcome(final_outcome(), now=NOW) self.assertEqual(normalized["verdict"], "succeeded") @@ -198,53 +214,21 @@ def test_experiment_gate_promotes_only_complete_held_out_evidence(self): evaluate_experiment(spec, invalid_chains, evaluated_at=NOW.isoformat()) def test_experiment_evaluation_is_persisted_and_replay_safe(self): - with tempfile.TemporaryDirectory() as root: - path = Path(root) - repository = path / "repo" - subprocess.run(["git", "init", "-q", str(repository)], check=True) - subprocess.run( - ["git", "-C", str(repository), "config", "user.email", "test@example.invalid"], - check=True, - ) - subprocess.run( - ["git", "-C", str(repository), "config", "user.name", "Test"], - check=True, - ) - (repository / "README").write_text("fixture\n") - subprocess.run(["git", "-C", str(repository), "add", "README"], check=True) - subprocess.run(["git", "-C", str(repository), "commit", "-qm", "base"], check=True) - store = Store(path / "state.sqlite3", path / "artifacts") - self.addCleanup(store.close) - spec = experiment(repository) - for case in spec["cases"]: - for arm, verdict in (("control", "failed"), ("candidate", "succeeded")): - outcome_id = case[f"{arm}_outcome_id"] - claim = store.claim_start( - repository, f"run-{outcome_id}", {}, "owner", - ) - store.connection.execute( - "UPDATE runs SET state=?,phase=NULL WHERE id=?", - (verdict, claim.run_id), - ) - store.record_outcome( - claim.run_id, - experimental_final(outcome_id, verdict), - now=NOW, - ) - first = store.evaluate_learning_experiment(spec, now=NOW) - replay = store.evaluate_learning_experiment(spec, now=NOW) - self.assertEqual(first["evaluation"]["verdict"], "promotion_proposal") - self.assertFalse(first["replayed"]) - self.assertTrue(replay["replayed"]) - changed = copy.deepcopy(spec) - changed["hypothesis"] = "Mutated after evaluation." - with self.assertRaisesRegex(ConflictError, "different specification"): - store.evaluate_learning_experiment(changed, now=NOW) - inputs = store.learning_proposal_inputs(repository, now=NOW) - self.assertEqual( - inputs["experiment"]["experiment"]["experiment_id"], - spec["experiment_id"], - ) + fixture = self.runtime_fixture() + first = fixture.service.policy_evaluate(fixture.spec) + replay = fixture.service.policy_evaluate(fixture.spec) + self.assertEqual(first["evaluation"]["verdict"], "promotion_proposal") + self.assertFalse(first["replayed"]) + self.assertTrue(replay["replayed"]) + self.assertTrue(first["eligibility"]["eligible"]) + changed = copy.deepcopy(fixture.spec) + changed["hypothesis"] = "Mutated after evaluation." + with self.assertRaisesRegex(ConflictError, "different specification"): + fixture.service.policy_evaluate(changed) + store = fixture.store() + self.addCleanup(store.close) + inputs = store.learning_proposal_inputs(fixture.repo) + self.assertEqual(inputs["experiment"]["experiment"]["experiment_id"], fixture.spec["experiment_id"]) def test_learning_proposal_is_traceable_and_never_changes_policy(self): project_path = "/tmp/experiment-project" @@ -267,35 +251,15 @@ def test_learning_proposal_is_traceable_and_never_changes_policy(self): {"action": "retain_current_policy", "review_required": False}, ) - spec = experiment(Path(project_path)) - chains = {} - for case in spec["cases"]: - chains[case["control_outcome_id"]] = { - "final": experimental_final(case["control_outcome_id"], "failed"), - "late_corrections": [], - } - chains[case["candidate_outcome_id"]] = { - "final": experimental_final(case["candidate_outcome_id"], "succeeded"), - "late_corrections": [], - } - evaluation = evaluate_experiment( - spec, chains, evaluated_at=NOW.isoformat(), - ) - spec_sha256 = hashlib.sha256( - canonical_json(validate_experiment(spec)).encode(), - ).hexdigest() - evaluation_sha256 = hashlib.sha256( - canonical_json(evaluation).encode(), - ).hexdigest() - record = { - "experiment": spec, - "spec_sha256": spec_sha256, - "evaluation": evaluation, - "evaluation_sha256": evaluation_sha256, - "recorded_at": NOW.isoformat(), - } + fixture = self.runtime_fixture() + evaluation = fixture.service.policy_evaluate(fixture.spec) + evaluation_sha256 = evaluation["evaluation_sha256"] + store = fixture.store() + self.addCleanup(store.close) + inputs = store.learning_proposal_inputs(fixture.repo) + report, record = inputs["report"], inputs["experiment"] proposal = build_learning_proposal( - report, record, generated_at=NOW.isoformat(), + report, record, generated_at=datetime.now(timezone.utc).isoformat(), ) self.assertEqual(proposal["verdict"], "promotion_proposal") self.assertFalse(proposal["active_policy_changed"]) @@ -429,7 +393,7 @@ def test_current_schema_contains_outcome_ledger(self): store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0], - 14, + 15, ) columns = { row[1] for row in store.connection.execute("PRAGMA table_info(outcomes)") diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index f356664..1d7be20 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -24,7 +24,7 @@ from devsquad.store import ConflictError, Store -NOW = datetime(2026, 9, 29, 9, 0, tzinfo=timezone.utc) +NOW = datetime.now(timezone.utc) def profile(profile_id, model_id, *, tools=None, pool="pool-a", billing="subscription"): @@ -159,8 +159,11 @@ def setUp(self): self.store = Store(self.database, self.artifacts) self.incumbent = profile("profile-a", "model-a") self.candidate = profile("profile-b", "model-b") + self.runtime_fixtures = [] def tearDown(self): + for fixture in self.runtime_fixtures: + fixture.close() self.store.close() self.temp.cleanup() @@ -170,66 +173,17 @@ def seed_experiment( experiment_id="experiment-profile-b", candidate_succeeds=True, ): - suffix = experiment_id.replace("experiment-", "") - cases = [ - { - "case_id": "eval-1", "split": "evaluation", - "control_outcome_id": f"control-eval-{suffix}", - "candidate_outcome_id": f"candidate-eval-{suffix}", - }, - { - "case_id": "hold-1", "split": "held_out", - "control_outcome_id": f"control-hold-{suffix}", - "candidate_outcome_id": f"candidate-hold-{suffix}", - }, - ] - for case in cases: - verdicts = ( - (("control", "failed"), ("candidate", "succeeded")) - if candidate_succeeds - else (("control", "succeeded"), ("candidate", "failed")) - ) - for arm, verdict in verdicts: - outcome_id = case[f"{arm}_outcome_id"] - claim = self.store.claim_start( - self.repo, f"run-{outcome_id}", {}, "owner", - ) - self.store.connection.execute( - "UPDATE runs SET state=?,phase=NULL WHERE id=?", - (verdict, claim.run_id), - ) - self.store.record_outcome( - claim.run_id, final_outcome(outcome_id, verdict), now=NOW, - ) - spec = { - "schema_version": 1, - "experiment_id": experiment_id, - "project_path": str(self.repo), - "question": "Should profile B replace profile A?", - "hypothesis": "Profile B improves held-out success.", - "evidence_availability": "tracked_fixture", - "variable": { - "kind": "profile_binding", - "alias": "review.deep", - "control_profile_id": "profile-a", - "candidate_profile_id": "profile-b", - }, - "cases": cases, - "gate": { - "min_evaluation_pairs": 1, - "min_held_out_pairs": 1, - "noninferiority_margin": 0.0, - "minimum_success_gain": 1.0, - "max_candidate_escaped_defects": 0, - }, - "budget": { - "max_cases": 2, - "max_worker_invocations": 0, - "wall_seconds": 60, - }, - "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, - } - return self.store.evaluate_learning_experiment(spec, now=NOW) + # Positive authority comes from real workers and public completion, + # never SQL-terminalized empty runs. The assignment seam remains R5. + from experiment_runtime_fixture import ExperimentRuntimeFixture + + fixture = ExperimentRuntimeFixture( + self.root / experiment_id, service=Service(self.root), repo=self.repo, + experiment_id=experiment_id, candidate_succeeds=candidate_succeeds, + ) + self.runtime_fixtures.append(fixture) + fixture.run_all() + return fixture.service.policy_evaluate(fixture.spec) def qualification(self, evaluation, *, qualification_id="qualification-b"): return { @@ -249,8 +203,8 @@ def qualification(self, evaluation, *, qualification_id="qualification-b"): "budget": { "max_cases": 2, "used_cases": 2, - "max_worker_invocations": 0, - "worker_invocations": 0, + "max_worker_invocations": 4, + "worker_invocations": 4, "max_wall_seconds": 60, "wall_seconds": 5, }, @@ -598,7 +552,7 @@ def test_schema_twelve_contains_lifecycle_ledger(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 14) + self.assertEqual(version, 15) tables = { row[0] for row in self.store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_store.py b/test/core/test_store.py index ef4af0e..d89e5c2 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -482,8 +482,8 @@ def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): ) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 14) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(15,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(16,'future')") self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -498,7 +498,7 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 14) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") @@ -515,7 +515,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],14) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],15) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From 5de66550d78e5e74a8df21a59bd202db24573c1e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 14:31:29 -0700 Subject: [PATCH 144/197] WIP checkpoint: Prove stale lifecycle and historical compatibility through public offline runs (2026-10-01 14:31) --- docs/plans/engineering-team/CONTRACTS.md | 5 +- docs/plans/engineering-team/RESUME.md | 27 +- docs/plans/engineering-team/backlog.json | 2 +- .../R3b2-public-compatibility-2026-10-01.json | 35 +++ .../src/devsquad/experiment_eligibility.py | 6 + plugin/core/src/devsquad/store.py | 11 + test/core/experiment_runtime_fixture.py | 58 +++- test/core/test_cli.py | 16 + test/core/test_experiment_eligibility.py | 275 +++++++++++++++++- test/core/test_experiment_saved_runs.py | 23 ++ 10 files changed, 435 insertions(+), 23 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index 11290d7..c67cedc 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -322,7 +322,10 @@ evaluation SHA256, which identifies either the original or reviewed revision. New qualifications must match the full tested candidate fingerprint and all assigned candidate tasks' declared role/task class. Qualification replay and every new binding mutation recheck current evidence within their write -transaction. A proven bootstrap predecessor without qualification retains its +transaction. A new promotion must compare the tested control fingerprint with +the current incumbent, not merely find a candidate that passed against some +other profile. Regression rollback requires complete evaluation and held-out +pairs and both exact tested fingerprints. A proven bootstrap predecessor without qualification retains its existing explicit baseline contract. Installation of this schema remains gated on old active/recoverable-run upgrade safety; adding the migration does not establish that installation gate. Implementation status is in RESUME.md. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 2fe866f..7183d9c 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -27,12 +27,27 @@ passed. No full integration gate or independent audit has run on schema 15. The source checkpoints, failure history and limitations are recorded in [R3b.2 partial evidence](evidence/R3b2-eligibility-partial-2026-10-01.json). -**Exact next action:** finish the negative/public proof for stale catalog -fallback and regression rollback, correction/qualification transaction races, -revision-chain/CLI semantics and upgraded historical duplicate-outcome report/ -proposal readability. Audit the shared gate and finish R3c's remaining public -coverage, then run one full unchanged-source integration gate at the coherent -R3 boundary. Preserve the original evaluations and completed decisions. Do +The next compatibility slice now proves stale regression rejection followed +by an explicit valid rollback review, stale qualified rollback-target rejection, +catalog fallback skipping that target (or blocking without any eligible +predecessor), correction/qualification writer fencing, current-versus-stale +proposal output and schema-13 duplicated-outcome history remaining readable +but ineligible. Fresh reviewed qualifying evidence pins a new promotion to its +revision hash. CLI revision dispatch passes. Four public delivery arms each +executed implementation, independent fixture review and checks with unchanged +source HEAD/checkout. Failures in that new test helper (wrong fixture shape +and treating a candidate-ready status as final review readiness) were repaired; +they are not represented as runtime defects or passing gates. + +The final targeted revision/installed-wheel checks passed two tests in 8.102 +seconds; the packaged wheel contains/applies schema 15. All 227 Bash assertions +and generated-reference/whitespace checks passed. This does not prove an old +active daemon can survive a schema change. + +**Exact next action:** run one full unchanged-source integration gate on the checkpointed schema-15 +source. No full gate on this source or independent audit is claimed yet. +The shared gate audit and bounded independent review remain required for R3 +closure. Preserve the original evaluations and completed decisions. Do not install schema 15 before the old active/recoverable-run upgrade test; R4–R6 and R8's Claude/Grok/Gemini installed proofs remain pending. No provider call, installation refresh, purchase, global setting change or push occurred. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 3781143..a439ce0 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -21,7 +21,7 @@ "status": "partial", "artifact": "evidence/R3b2-eligibility-partial-2026-10-01.json", "scope": "Reader full gate passed at a4a87fd; schema-15 explicit revisions and shared lifecycle eligibility implemented with focused public fixtures; no install or provider calls", - "next_action": "Finish stale fallback/rollback, transaction race, revision/CLI and upgraded legacy public proof; audit R3 and run its complete integration gate", + "next_action": "Stale fallback/rollback, race, CLI, upgraded legacy and real delivery-pair proofs pass; finish revision/wheel checks, audit R3 and run its complete integration gate", "limitations": "Schema-15 full integration and independent audit incomplete; SQLite warning and production paired-trial controller remain R5; native Gemini worker and refreshed installed proof unverified" }, "planning_checkpoint": { diff --git a/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json new file mode 100644 index 0000000..548d19d --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json @@ -0,0 +1,35 @@ +{ + "schema_version": 1, + "work_package": "R3b.2/R3c", + "status": "public_compatibility_verified_full_integration_pending", + "recorded_at": "2026-10-01T21:29:24Z", + "baseline_revision": "4e6297f", + "source_sha256": { + "plugin/core/src/devsquad/experiment_eligibility.py": "66d1c4d079e4824daf5cd349fa03a35f09b70d2b54367447b2633092d60a5baa", + "plugin/core/src/devsquad/store.py": "881398323a238d0f5538c6ce2cfeb8b9dc630a5b617ee5c2c794597c1099c8c6", + "test/core/experiment_runtime_fixture.py": "cf98bc312665b138e4d4d9a9633f888ef2f2e48ecba55a56037103402afa7646", + "test/core/test_experiment_eligibility.py": "940d6fb19cf9baa5685c633094b2b309b261504f5546785acee6c227b453f897", + "test/core/test_experiment_saved_runs.py": "68ce3265c5c342805dbf433760172d2b531694b821421e3ee1f9d1d24bba9240" + }, + "verification": { + "stale_rollback_and_fallback": "2 tests in 26.432 seconds passed: stale regression blocked, explicit revision permitted valid rollback; stale qualified target blocked rollback and was skipped by catalog fallback; proven bootstrap preserved", + "race_and_proposal": "Both targeted tests passed during a 3-test 11.138-second run; the third historical fixture test failed, so that combined run is not a pass", + "historical_failure": "Historical fixture stored /var project path instead of resolved /private/var path and proposal returned no_evaluated_experiment; corrected the fixture to match old Store normalization", + "historical_green": "Schema-13 duplicate-outcome upgrade/report/proposal/new-authority rejection test passed separately in 0.074 seconds", + "cli": "Original and explicit-revision dispatch tests passed, 2 tests in 0.006 seconds", + "delivery_fixture_failures": "New fixture first used an invalid internal implementation shape (0.555 seconds), then waited on an internal phase (20.605 seconds), then mistook pre-daemon candidate readiness for reviewed completion (1.426 seconds). These were test-helper defects, not runtime findings", + "delivery_green": "Four public paired issue-delivery arms with 8 real worker attempts, mandatory checks and source preservation passed in 7.988 seconds", + "fresh_reviewed_qualification": "1 test in 5.449 seconds passed; fresh reviewed revision qualifies and pins the new binding decision", + "revision_and_wheel": "2 tests in 8.102 seconds passed: revision arguments/predecessor checks and installed-wheel current schema-15 content/migration", + "bash": "227 assertions in 11 files passed", + "generated_reference_and_whitespace": "Both checks passed", + "full_core": "Pending on this source; do not substitute earlier R3b.1 or partial source results", + "independent_audit": "Not completed" + }, + "remaining": [ + "Complete unchanged-source core integration", + "Bounded independent R3 audit", + "R4-R6 runtime repairs and R8 safe updated installation/live Claude, Grok and Gemini proofs" + ], + "external_actions": "No provider request, installation refresh, purchase, reset, global settings change or push" +} diff --git a/plugin/core/src/devsquad/experiment_eligibility.py b/plugin/core/src/devsquad/experiment_eligibility.py index 6e81ef4..6b61c36 100644 --- a/plugin/core/src/devsquad/experiment_eligibility.py +++ b/plugin/core/src/devsquad/experiment_eligibility.py @@ -37,6 +37,12 @@ def saved_evaluation( (experiment_id, evaluation_sha256), ).fetchone() if revision is not None: + parent = revision["previous_evaluation_sha256"] + if parent != original["evaluation_sha256"] and connection.execute( + "SELECT 1 FROM experiment_evaluation_revisions WHERE experiment_id=? " + "AND evaluation_sha256=? AND id Date: Thu, 1 Oct 2026 14:41:25 -0700 Subject: [PATCH 145/197] WIP checkpoint: Record failed integration and close installer test SQLite connection (2026-10-01 14:41) --- docs/plans/engineering-team/RESUME.md | 18 ++++++++++++++++-- .../R3b2-public-compatibility-2026-10-01.json | 4 +++- test/core/test_check_integrity_runtime.py | 11 ++++++++--- test/core/test_install_core.py | 3 ++- 4 files changed, 29 insertions(+), 7 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 7183d9c..2e4f22b 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -44,8 +44,22 @@ seconds; the packaged wheel contains/applies schema 15. All 227 Bash assertions and generated-reference/whitespace checks passed. This does not prove an old active daemon can survive a schema change. -**Exact next action:** run one full unchanged-source integration gate on the checkpointed schema-15 -source. No full gate on this source or independent audit is claimed yet. +The full unchanged-source run at `5de6655` completed **441 tests in 372.506 +seconds, FAILED with one error and two optional-SDK skips**. The headless +delivery mutation test could not find `receipt.json`; its temporary runtime +was cleaned before diagnosis. That case passes unchanged (1 test, 2.760 +seconds), and all nine check-integrity tests pass together (19.194 seconds). +This does not erase the failed integration gate or establish its cause. The +test now reports saved status/artifact names if the missing receipt recurs. +The SQLite finalizer warning has a concrete test-owned candidate: the installer +survival test used a SQLite transaction context without closing the connection. +It now uses `contextlib.closing`; its targeted active-release test passes in +6.837 seconds with ResourceWarning promoted to error, forced collection and +zero unraisable exceptions. The complete suite must still confirm no warning. + +**Exact next action:** run one full unchanged-source integration gate on the +checkpointed schema-15 source. All 227 Bash assertions pass. No +passing full gate on this source or independent audit is claimed yet. The shared gate audit and bounded independent review remain required for R3 closure. Preserve the original evaluations and completed decisions. Do not install schema 15 before the old active/recoverable-run upgrade test; R4–R6 diff --git a/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json index 548d19d..828075b 100644 --- a/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json @@ -23,7 +23,9 @@ "revision_and_wheel": "2 tests in 8.102 seconds passed: revision arguments/predecessor checks and installed-wheel current schema-15 content/migration", "bash": "227 assertions in 11 files passed", "generated_reference_and_whitespace": "Both checks passed", - "full_core": "Pending on this source; do not substitute earlier R3b.1 or partial source results", + "full_core": "At 5de6655: 441 tests in 372.506 seconds FAILED (1 missing-receipt error, 2 optional-SDK skips); UTC elapsed 372.689586 and monotonic 372.691304 seconds. Not a passing gate", + "failed_case_recheck": "The exact headless delivery check-integrity case passed unchanged (1 test in 2.760 seconds); all 9 check-integrity cases passed in 19.194 seconds. Cause remains unproven; added status/artifact-name diagnostics", + "sqlite_cleanup": "Found installer test transaction context that did not close its SQLite connection; replaced with contextlib.closing. Active-release test passed in 6.837 seconds under ResourceWarning=error, forced collection and zero unraisable exceptions; complete-suite warning recheck pending", "independent_audit": "Not completed" }, "remaining": [ diff --git a/test/core/test_check_integrity_runtime.py b/test/core/test_check_integrity_runtime.py index 448632a..486843b 100644 --- a/test/core/test_check_integrity_runtime.py +++ b/test/core/test_check_integrity_runtime.py @@ -159,10 +159,15 @@ def decision(packet, submission_id, disposition): return {**body, "submission_hash": request_hash(body)} def receipt(self, run_id): - artifact = next( - item for item in self.service.result(run_id)["artifacts"] + result = self.service.result(run_id) + artifact = next(( + item for item in result["artifacts"] if item["name"] == "receipt.json" - ) + ), None) + self.assertIsNotNone(artifact, { + "status": self.service.status(run_id), + "artifact_names": [item["name"] for item in result["artifacts"]], + }) return json.loads(Path(artifact["path"]).read_text()) def snapshot(self, run_id): diff --git a/test/core/test_install_core.py b/test/core/test_install_core.py index e7f1cb3..a640371 100644 --- a/test/core/test_install_core.py +++ b/test/core/test_install_core.py @@ -1,4 +1,5 @@ import json +from contextlib import closing import os from pathlib import Path import re @@ -426,7 +427,7 @@ def test_update_does_not_break_an_active_release_pinned_run(self): launcher, "result", started["run_id"], "--runtime-dir", str(runtime), ) self.assertTrue(result["data"]["ready"]) - with sqlite3.connect(runtime / "state.sqlite3") as connection: + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: package_path, package_digest = connection.execute( "SELECT package_path, package_digest FROM runs WHERE id=?", (started["run_id"],), From f1210003930b8df0cdef9e185fa4283a5fbf36ae Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 14:48:59 -0700 Subject: [PATCH 146/197] WIP checkpoint: Record schema-15 full integration pass and retained failure history (2026-10-01 14:48) --- docs/plans/engineering-team/RESUME.md | 17 ++++++++++++++--- docs/plans/engineering-team/backlog.json | 4 ++-- .../R3b2-public-compatibility-2026-10-01.json | 6 +++--- 3 files changed, 19 insertions(+), 8 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 2e4f22b..8a943d0 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -57,9 +57,20 @@ It now uses `contextlib.closing`; its targeted active-release test passes in 6.837 seconds with ResourceWarning promoted to error, forced collection and zero unraisable exceptions. The complete suite must still confirm no warning. -**Exact next action:** run one full unchanged-source integration gate on the -checkpointed schema-15 source. All 227 Bash assertions pass. No -passing full gate on this source or independent audit is claimed yet. +The unchanged-source rerun at `dc4e110` completed **441 tests in 374.998 +seconds, OK with two optional-SDK skips**, with forced collection and **zero +unraisable exceptions** (no SQLite warning). UTC/monotonic elapsed times agreed +at 375.115/375.116 seconds. Retain the earlier failed run; its missing-receipt +cause remains unproven, not represented as a source repair. All 227 Bash +assertions and generated reference passed before this checkpoint. + +**Exact next action:** finish bounded R3 independent audit and continue R4's +normal routing/catalog/quota connections. Native Codex discovery currently +fails closed: PATH is 0.135.0, the previously verified bundled executable moved +to `ChatGPT.app/Contents/Resources/codex-cli/bin/codex` and is now 0.159.2. +A non-generating 0.159.2 probe passed initialize, complete model/list, +account/read and rateLimits/read without settings changes or model requests. +Version/path compatibility and operation verification remain separate gates. The shared gate audit and bounded independent review remain required for R3 closure. Preserve the original evaluations and completed decisions. Do not install schema 15 before the old active/recoverable-run upgrade test; R4–R6 diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index a439ce0..c0e719a 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -21,8 +21,8 @@ "status": "partial", "artifact": "evidence/R3b2-eligibility-partial-2026-10-01.json", "scope": "Reader full gate passed at a4a87fd; schema-15 explicit revisions and shared lifecycle eligibility implemented with focused public fixtures; no install or provider calls", - "next_action": "Stale fallback/rollback, race, CLI, upgraded legacy and real delivery-pair proofs pass; finish revision/wheel checks, audit R3 and run its complete integration gate", - "limitations": "Schema-15 full integration and independent audit incomplete; SQLite warning and production paired-trial controller remain R5; native Gemini worker and refreshed installed proof unverified" + "next_action": "441-test schema-15 integration passes at dc4e110 with two optional-SDK skips and zero unraisable exceptions; finish bounded independent R3 audit and R4 routing/catalog/quota", + "limitations": "Independent R3 audit incomplete; production paired-trial controller remains R5; Codex bundled path/version changed to 0.159.2 and needs compatibility update; native Gemini worker and refreshed installed proof unverified" }, "planning_checkpoint": { "recorded_on": "2026-10-01", diff --git a/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json b/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json index 828075b..d7ef154 100644 --- a/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R3b2-public-compatibility-2026-10-01.json @@ -1,7 +1,7 @@ { "schema_version": 1, "work_package": "R3b.2/R3c", - "status": "public_compatibility_verified_full_integration_pending", + "status": "source_offline_verified_independent_audit_pending", "recorded_at": "2026-10-01T21:29:24Z", "baseline_revision": "4e6297f", "source_sha256": { @@ -25,11 +25,11 @@ "generated_reference_and_whitespace": "Both checks passed", "full_core": "At 5de6655: 441 tests in 372.506 seconds FAILED (1 missing-receipt error, 2 optional-SDK skips); UTC elapsed 372.689586 and monotonic 372.691304 seconds. Not a passing gate", "failed_case_recheck": "The exact headless delivery check-integrity case passed unchanged (1 test in 2.760 seconds); all 9 check-integrity cases passed in 19.194 seconds. Cause remains unproven; added status/artifact-name diagnostics", - "sqlite_cleanup": "Found installer test transaction context that did not close its SQLite connection; replaced with contextlib.closing. Active-release test passed in 6.837 seconds under ResourceWarning=error, forced collection and zero unraisable exceptions; complete-suite warning recheck pending", + "sqlite_cleanup": "Found installer test transaction context that did not close its SQLite connection; replaced with contextlib.closing. Active-release test passed in 6.837 seconds under ResourceWarning=error, forced collection and zero unraisable exceptions. Full 441-test rerun at dc4e110 had zero unraisable exceptions and no SQLite warning", + "full_core_rerun": "At dc4e110: 441 tests in 374.998 seconds OK (2 optional-SDK skips); UTC elapsed 375.114707 and monotonic 375.116386 seconds; forced collection, zero unraisable exceptions. Earlier failed gate retained", "independent_audit": "Not completed" }, "remaining": [ - "Complete unchanged-source core integration", "Bounded independent R3 audit", "R4-R6 runtime repairs and R8 safe updated installation/live Claude, Grok and Gemini proofs" ], From 933419041fc8aa415dda622810e8a17fde41aac5 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 14:50:54 -0700 Subject: [PATCH 147/197] WIP checkpoint: Support the current bundled Codex layout and retain discovery evidence (2026-10-01 14:50) --- docs/plans/engineering-team/RESUME.md | 8 +++++++ .../R4-native-compatibility-2026-10-01.json | 24 +++++++++++++++++++ plugin/core/adapters/codex/adapter.json | 4 ++-- .../core/integrations/codex/registration.json | 2 +- test/core/test_m1.py | 24 +++++++++++++++++++ 5 files changed, 59 insertions(+), 3 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R4-native-compatibility-2026-10-01.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 8a943d0..8c9e489 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -71,6 +71,14 @@ to `ChatGPT.app/Contents/Resources/codex-cli/bin/codex` and is now 0.159.2. A non-generating 0.159.2 probe passed initialize, complete model/list, account/read and rateLimits/read without settings changes or model requests. Version/path compatibility and operation verification remain separate gates. +The new bundled layout/version now has an explicit manifest and registration +entry. Its regression first failed on the missing path; 61 M1/MCP/task-entry +tests then passed (two optional-SDK skips), and all 227 Bash assertions passed. +Before the next source slice, checkpoint and run one bounded, source-only +native Codex R3 audit against the frozen `672e383` → current range in a private +runtime. This is not an updated-installation or Claude-delivery proof. Retain +the exact reviewer/check artifacts and any findings; do not claim completion +without reviewing them. No paid fallback or global configuration change. The shared gate audit and bounded independent review remain required for R3 closure. Preserve the original evaluations and completed decisions. Do not install schema 15 before the old active/recoverable-run upgrade test; R4–R6 diff --git a/docs/plans/engineering-team/evidence/R4-native-compatibility-2026-10-01.json b/docs/plans/engineering-team/evidence/R4-native-compatibility-2026-10-01.json new file mode 100644 index 0000000..f3b0c1f --- /dev/null +++ b/docs/plans/engineering-team/evidence/R4-native-compatibility-2026-10-01.json @@ -0,0 +1,24 @@ +{ + "schema_version": 1, + "work_package": "R4/R8", + "status": "discovery_compatibility_verified_operation_pending", + "observations": { + "path_cli": "codex-cli 0.135.0 is unverified and is not selected for native work", + "bundled_binary": "/Applications/ChatGPT.app/Contents/Resources/codex-cli/bin/codex", + "bundled_version": "codex-cli 0.159.2", + "non_generating_protocol": "initialize, complete model/list pagination, account/read refreshToken=false and account/rateLimits/read all succeeded", + "model_catalog": "8 models with native effort metadata, one default", + "auth": "chatgpt; no credentials or account identifier retained here", + "protocol_schema": "Generated by the installed bundled binary; official account/rateLimits documentation also inspected" + }, + "verification": { + "red": "New bundled layout/version test failed on missing manifest path before the repair", + "green": "61 M1/MCP/task-entry tests passed in 5.886 seconds, with two optional-SDK skips", + "bash": "227 assertions in 11 files passed", + "native_generation": "Not run for this metadata slice", + "installation": "Not updated" + }, + "documentation": "https://learn.chatgpt.com/docs/app-server", + "remaining": ["Independent R3 audit", "R4 stable aliases/scoped catalog/quota", "R8 safe upgrade and installed provider proofs"], + "external_mutations": "No global settings, authentication changes, purchases, credit resets, paid APIs, pushes or installation refresh" +} diff --git a/plugin/core/adapters/codex/adapter.json b/plugin/core/adapters/codex/adapter.json index 9fdc58a..73f33e7 100644 --- a/plugin/core/adapters/codex/adapter.json +++ b/plugin/core/adapters/codex/adapter.json @@ -3,9 +3,9 @@ "name": "codex", "transport": "native_protocol", "fallback_transport": "cli_exec", - "binary_candidates": ["codex", "/Applications/ChatGPT.app/Contents/Resources/codex", "/Applications/Codex.app/Contents/Resources/codex"], + "binary_candidates": ["codex", "/Applications/ChatGPT.app/Contents/Resources/codex-cli/bin/codex", "/Applications/ChatGPT.app/Contents/Resources/codex", "/Applications/Codex.app/Contents/Resources/codex"], "model_provider": "openai", - "verified_harness_versions": ["codex-cli 0.153.4", "codex-cli 0.155.0-alpha.9.2"], + "verified_harness_versions": ["codex-cli 0.153.4", "codex-cli 0.155.0-alpha.9.2", "codex-cli 0.159.2"], "capabilities": {"efforts_by_model": {}, "native_model_list": true, "resume": true}, "permission_profiles": {"read_only": [], "workspace_write": []}, "output_format": "jsonl" diff --git a/plugin/core/integrations/codex/registration.json b/plugin/core/integrations/codex/registration.json index 1372b3b..0651cf1 100644 --- a/plugin/core/integrations/codex/registration.json +++ b/plugin/core/integrations/codex/registration.json @@ -2,7 +2,7 @@ "schema_version": 1, "id": "codex", "display_name": "Codex", - "executable_paths": ["/Applications/ChatGPT.app/Contents/Resources/codex", "/Applications/Codex.app/Contents/Resources/codex"], + "executable_paths": ["/Applications/ChatGPT.app/Contents/Resources/codex-cli/bin/codex", "/Applications/ChatGPT.app/Contents/Resources/codex", "/Applications/Codex.app/Contents/Resources/codex"], "executable_names": ["codex"], "server_name": "devsquad", "surface": "codex-app", diff --git a/test/core/test_m1.py b/test/core/test_m1.py index 951d14a..4e39195 100644 --- a/test/core/test_m1.py +++ b/test/core/test_m1.py @@ -56,6 +56,30 @@ def test_verified_binary_candidate_wins_over_an_older_path_binary(self): with patch.dict(os.environ, {"PATH": str(older.parent)}): self.assertEqual(manifest.resolve_binary(), str(verified)) + def test_current_bundled_layout_and_version_are_explicitly_supported(self): + manifest = self.manifest("codex") + bundled = "/Applications/ChatGPT.app/Contents/Resources/codex-cli/bin/codex" + self.assertIn(bundled, manifest.binary_candidates) + self.assertIn("codex-cli 0.159.2", manifest.verified_versions) + registration = json.loads((CORE / "integrations/codex/registration.json").read_text()) + self.assertIn(bundled, registration["executable_paths"]) + temp, older = self.fake_path("codex") + self.addCleanup(temp.cleanup) + older.write_text("#!/bin/sh\necho 'codex-cli 0.135.0'\n") + current = Path(temp.name) / "current-codex" + current.write_text("#!/bin/sh\necho 'codex-cli 0.159.2'\n") + current.chmod(0o700) + manifest = replace(manifest, binary_candidates=("codex", str(current))) + with patch.dict(os.environ, {"PATH": str(older.parent)}): + self.assertEqual(manifest.resolve_binary(), str(current)) + spec = prepare_native_codex( + manifest.with_model_efforts({"gpt-test": ("low",)}), + cwd=temp.name, model="gpt-test", effort="low", + permission="read_only", timeout_seconds=9, + harness_version_value="codex-cli 0.159.2", + ) + self.assertEqual(spec.requested.verification, "verified") + def test_prepare_preserves_spaces_and_tsx_prompt(self): temp, binary = self.fake_path("agy") self.addCleanup(temp.cleanup) From db55d3fd01d9cae601d363384260d1fbfec9042e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 15:01:25 -0700 Subject: [PATCH 148/197] WIP checkpoint: Preserve partial normal aliases and independent R3 audit findings (2026-10-01 15:01) --- docs/plans/engineering-team/RESUME.md | 31 +++++ .../R3-independent-audit-2026-10-01.json | 21 ++++ plugin/core/src/devsquad/cli.py | 30 +++-- plugin/core/src/devsquad/service.py | 35 ++++++ plugin/core/src/devsquad/task_entry.py | 104 +++++++++++----- test/core/experiment_runtime_fixture.py | 6 +- test/core/test_cli.py | 2 + test/core/test_task_entry.py | 117 ++++++++++++++++-- 8 files changed, 295 insertions(+), 51 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R3-independent-audit-2026-10-01.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 8c9e489..f0edbae 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -79,6 +79,37 @@ native Codex R3 audit against the frozen `672e383` → current range in a privat runtime. This is not an updated-installation or Claude-delivery proof. Retain the exact reviewer/check artifacts and any findings; do not claim completion without reviewing them. No paid fallback or global configuration change. + +### Independent R3 audit — findings open; R4 alias slice partial + +The real read-only Codex 0.159.2 / `gpt-6.1-sol` high review completed against +`672e383` → `9334190` in private runtime +`/Users/Dikshant/.devsquad/private-probes/r3-independent-audit-20261001`, run +`80fdd336-07e2-445c-8a08-5877cd6534df`. It reports **R3-001 (high)**: delivery +implementation/reviewer imports are optional and stdout is not semantically +validated; **R3-002 (medium)**: submitted latency/usage ratios are not bound to +saved measurements. Both require reproductions and repairs. The normal Bash +check exited 0 but was correctly invalidated after creating undeclared Python +bytecode. The host rejected the packet and the run is terminal `failed`, version +22. No clean audit or installed proof is claimed. Native usage was reported +as 1,642,079 input / 10,733 output tokens, one worker invocation; internal model +request count is unknown. Do not repeat this broad audit. Re-review only the +bounded repairs when ready. + +In-flight R4 work adds normal stable aliases, explicit pin provenance and +policy-matched incumbent lookup before discovery. Two red tests reproduced the +old concrete-role bypass/missing binding API. The 29-test task-entry/CLI gate +passes in 5.073 seconds. A new public promotion-to-normal-entry fixture fails +early because its synthetic declaration lacks native adapter evidence for its +Codex profiles; this is a test-helper issue, not a passing public chain. The +new fixture parameters and test are deliberately checkpointed as **partial**; +do not weaken native experiment provenance to make them pass. + +**Exact next action:** repair the two independent R3 findings, with truthful +delivery success/failure evidence and measured-or-unknown ratio regressions; +fix check bytecode generation without weakening candidate integrity. Then +finish R4's public normal-entry proof, scoped catalog/quota, R5/R6 and the safe +R8 installation/Claude/Grok/Gemini gates. Installed release remains unchanged. The shared gate audit and bounded independent review remain required for R3 closure. Preserve the original evaluations and completed decisions. Do not install schema 15 before the old active/recoverable-run upgrade test; R4–R6 diff --git a/docs/plans/engineering-team/evidence/R3-independent-audit-2026-10-01.json b/docs/plans/engineering-team/evidence/R3-independent-audit-2026-10-01.json new file mode 100644 index 0000000..864b779 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3-independent-audit-2026-10-01.json @@ -0,0 +1,21 @@ +{ + "schema_version": 1, + "work_package": "R3", + "status": "findings_open", + "base_revision": "672e38305fcc764d87d7da6afb5914cbf3c5a3ce", + "target_revision": "933419041fc8aa415dda622810e8a17fde41aac5", + "run_id": "80fdd336-07e2-445c-8a08-5877cd6534df", + "observed_identity": {"harness": "codex", "version": "codex-cli 0.159.2", "model": "gpt-6.1-sol", "effort": "high", "permission": "read_only", "verification": "verified"}, + "review_sha256": "97c4d5ccd0874fc7eaaf3a24311cb6edbba309c5abdb8ba714b517bafc8fdba8", + "attempt_sha256": "1c2b381acf2961bb96557a81b53554fb5dba8e0c7bd7965d8cbf0eff975c720b", + "findings": [ + {"id": "R3-001", "severity": "high", "path": "plugin/core/src/devsquad/experiment_evidence.py", "summary": "Delivery exposure can be accepted without semantically validated implementation/reviewer output and matching import or terminal failure evidence", "status": "open"}, + {"id": "R3-002", "severity": "medium", "path": "plugin/core/src/devsquad/store.py", "summary": "Latency/usage qualification ratios are trusted from the caller instead of verified saved measurements", "status": "open"} + ], + "checks": {"diff": "passed", "bash": "Exited 0, but invalidated by undeclared Python bytecode in the check workspace; candidate integrity correctly blocked acceptance"}, + "host_disposition": "reject", + "terminal": {"state": "failed", "version": 22}, + "native_usage": {"input_tokens": 1642079, "output_tokens": 10733, "source": "native_reported", "worker_invocations": 1, "native_model_requests": null}, + "limitations": "Source-only audit, not an updated-installation or Claude-to-Codex delivery proof. Full integration pass precedes these findings and does not close them", + "next_action": "Reproduce and repair the two bounded findings and bytecode/check friction. Do not repeat the broad review; re-review only repaired boundaries" +} diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index a8c7c9a..490028e 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -215,14 +215,21 @@ def _command_normal_entry( if args.dry_run and args.wait: raise ContractError("--wait cannot be combined with --dry-run") repo = resolve_repository(args.project_dir) + reviewer_model = args.model if workflow == "branch-review" else args.review_model + reviewer_effort = args.effort if workflow == "branch-review" else args.review_effort + pins = tuple(role for role, explicit in ( + ("reviewer", reviewer_model is not None or reviewer_effort is not None), + ("implementer", workflow == "issue-delivery" and (args.implementer_model is not None or args.implementer_effort is not None)), + ) if explicit) + bindings = ( + _service(args).normal_entry_bindings(workflow, pinned_roles=pins) + if (Path(args.runtime_dir) / "state.sqlite3").is_file() else {} + ) + incumbent = bindings.get("reviewer", {}).get("profile", {}) codex_identity = discover_codex_identity( repo, - requested_model=( - args.model if workflow == "branch-review" else args.review_model - ), - requested_effort=( - args.effort if workflow == "branch-review" else args.review_effort - ), + requested_model=reviewer_model or incumbent.get("model_id"), + requested_effort=reviewer_effort or incumbent.get("effort", {}).get("value"), ) if workflow == "branch-review": mode = args.mode @@ -240,8 +247,9 @@ def _command_normal_entry( focus = args.review_focus goal = args.issue write_paths = tuple(args.write_path or ()) - claude_model = args.implementer_model - claude_effort = args.implementer_effort + incumbent = bindings.get("implementer", {}).get("profile", {}) + claude_model = args.implementer_model or incumbent.get("model_id", "sonnet") + claude_effort = args.implementer_effort or incumbent.get("effort", {}).get("value", "high") task, summary = build_managed_task( workflow=workflow, project_dir=repo, @@ -256,6 +264,8 @@ def _command_normal_entry( review_focus=focus, claude_model=claude_model, claude_effort=claude_effort, + role_bindings=bindings, + pinned_roles=pins, ) idempotency_key = args.idempotency_key or ( f"normal-{workflow}-{summary['task_sha256']}" @@ -470,8 +480,8 @@ def parser() -> argparse.ArgumentParser: default="standard", ) fix.add_argument("--review-focus") - fix.add_argument("--implementer-model", default="sonnet") - fix.add_argument("--implementer-effort", default="high") + fix.add_argument("--implementer-model") + fix.add_argument("--implementer-effort") fix.add_argument("--idempotency-key") fix.add_argument("--dry-run", action="store_true") fix.add_argument("--wait", action="store_true") diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 616e68c..1b26676 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -75,6 +75,41 @@ def __init__(self, runtime: Path): def _store(self) -> Store: return Store(self.database, self.artifacts) + def normal_entry_bindings(self, workflow: str, *, pinned_roles: tuple[str, ...] = ()) -> dict[str, Any]: + """Read policy-matched approved incumbents before native discovery.""" + from .lifecycle import profile_fingerprint, profile_template_violation + from .task_entry import NORMAL_ALIASES, NORMAL_POLICY + + if workflow not in {"branch-review", "issue-delivery"}: + raise ContractError("normal entry workflow is unsupported") + task_class = "managed-review" if workflow == "branch-review" else "managed-fix" + roles = ("reviewer",) if workflow == "branch-review" else ("implementer", "reviewer") + store = self._store() + try: + store.connection.execute("BEGIN") + result = {} + for role in roles: + if role in pinned_roles: + continue + record = store.profile_binding(NORMAL_ALIASES[role]) + if record is None or record["template"]["policy"] != NORMAL_POLICY: + continue + template = record["template"] + profile = record["profile"] + if (profile_fingerprint(profile) != record["profile_sha256"] + or hashlib.sha256(canonical_json(template).encode()).hexdigest() != record["template_sha256"] + or profile_template_violation(profile, template) is not None + or task_class not in template["allowed_task_classes"] + or profile["quality_status"] != "proven"): + raise ContractError("normal alias incumbent is not approved for this task") + if record["qualification_id"] is not None: + store._require_current_qualification(record["qualification_id"], now=datetime.now(timezone.utc)) + result[role] = record + store.connection.execute("COMMIT") + return result + finally: + store.close() + def _preparation_failure_artifacts( self, store: Store, diff --git a/plugin/core/src/devsquad/task_entry.py b/plugin/core/src/devsquad/task_entry.py index 94836dc..75243d1 100644 --- a/plugin/core/src/devsquad/task_entry.py +++ b/plugin/core/src/devsquad/task_entry.py @@ -21,7 +21,7 @@ ) from .contracts import ContractError from .store import canonical_json -from .validation import validate_task +from .validation import validate_profile, validate_task SOURCE_ROOT = Path(__file__).resolve().parents[2] @@ -31,6 +31,8 @@ else Path(sys.prefix) / "share" / "devsquad" ) MAX_USER_CHECKS = 12 +NORMAL_POLICY = {"id": "managed-normal-entry", "version": 1} +NORMAL_ALIASES = {"implementer": "implement.balanced", "reviewer": "review.deep"} def _git(repo: Path, *arguments: str) -> str: @@ -193,6 +195,8 @@ def _managed_routing( *, claude_model: str, claude_effort: str, + role_bindings: dict[str, Any], + pinned_roles: set[str], ) -> dict[str, Any]: reviewer = { "id": "managed-codex-reviewer", @@ -202,29 +206,19 @@ def _managed_routing( "effort": {"value": codex["effort"], "transport": "native"}, "required_tools": ["read"], "permission_policy": "read_only", - "account_pool_id": "codex-subscription", + "account_pool_id": codex.get("account_pool_id", "codex-subscription"), "billing_mode": "subscription", "quality_status": "trial", "evidence_refs": [ f"runtime-catalog:{codex['harness_version']}:{codex['model_id']}", ], } - profiles = [reviewer] - roles: dict[str, list[dict[str, str]]] = { - "reviewer": [{"kind": "profile", "id": reviewer["id"]}], - } - account_pools: dict[str, dict[str, Any]] = { - "codex-subscription": { - "allowed_billing_modes": ["subscription"], - "max_concurrency": 1, - "unknown_capacity_policy": "allow_bounded", - }, - } + trial_profiles = {"reviewer": reviewer} if workflow == "issue-delivery": implementer = { "id": "managed-claude-implementer", "harness": "claude", - "model_family": "claude-sonnet", + "model_family": next((f"claude-{family}" for family in ("sonnet", "opus", "haiku") if family in claude_model.lower()), "claude"), "model_id": claude_model, "effort": {"value": claude_effort, "transport": "native"}, "required_tools": ["read", "write"], @@ -234,25 +228,57 @@ def _managed_routing( "quality_status": "trial", "evidence_refs": ["verified-claude-cli-2.1.220"], } - profiles.insert(0, implementer) - roles["implementer"] = [ - {"kind": "profile", "id": implementer["id"]}, - ] - account_pools["claude-subscription"] = { + trial_profiles = {"implementer": implementer, **trial_profiles} + profiles = [] + bindings = {} + overrides = {} + roles = {} + for role, profile in trial_profiles.items(): + # A catalog/default/pin change must not reuse a concrete profile ID + # with different bytes or overwrite an approved incumbent. + digest = hashlib.sha256(canonical_json(profile).encode()).hexdigest() + profile["id"] += f"-{digest[:16]}" + profiles.append(profile) + alias = NORMAL_ALIASES[role] + approved = role_bindings.get(role) + if approved is not None: + if (not isinstance(approved, dict) or approved.get("alias") != alias + or type(approved.get("version")) is not int or approved["version"] < 1): + raise ContractError("normal role binding identity is invalid") + incumbent = json.loads(canonical_json(approved.get("profile"))) + validate_profile(incumbent) + if (incumbent["harness"] != profile["harness"] + or incumbent["permission_policy"] != profile["permission_policy"] + or incumbent["billing_mode"] != "subscription" + or incumbent["quality_status"] != "proven"): + raise ContractError("normal role binding exceeds the supported role contract") + if incumbent["id"] == profile["id"] and incumbent != profile: + raise ContractError("normal role binding conflicts with the trial profile") + if incumbent["id"] != profile["id"]: + profiles.append(incumbent) + bindings[alias] = {"profile_id": incumbent["id"], "version": approved["version"]} + else: + bindings[alias] = {"profile_id": profile["id"], "version": 1} + roles[role] = [{"kind": "alias", "id": alias}] + if role in pinned_roles: + overrides[role] = {"profile_id": profile["id"], "fallback": "none"} + account_pools = { + profile["account_pool_id"]: { "allowed_billing_modes": ["subscription"], "max_concurrency": 1, "unknown_capacity_policy": "allow_bounded", } + for profile in profiles + } return { "profiles": { "schema_version": 1, "profiles": profiles, - "bindings": {}, + "bindings": bindings, }, "policy": { "schema_version": 1, - "id": "managed-normal-entry", - "version": 1, + **NORMAL_POLICY, "roles": roles, "task_classes": { "managed-review" if workflow == "branch-review" @@ -264,6 +290,7 @@ def _managed_routing( "experiment_budget": {}, "decision_helper": {"schema_version": 1, "mode": "off"}, }, + **({"overrides": overrides} if overrides else {}), } @@ -338,9 +365,14 @@ def build_managed_task( review_focus: str | None = None, claude_model: str = "sonnet", claude_effort: str = "high", + role_bindings: dict[str, Any] | None = None, + pinned_roles: Iterable[str] = (), ) -> tuple[dict[str, Any], dict[str, Any]]: if workflow not in {"branch-review", "issue-delivery"}: raise ContractError("normal entry workflow is unsupported") + if role_bindings is not None and not isinstance(role_bindings, dict): + raise ContractError("normal role bindings must be an object") + pins = set(pinned_roles) if not isinstance(goal, str) or not goal.strip(): raise ContractError("goal must be non-empty") if type(check_timeout) is not int or not 1 <= check_timeout <= 3600: @@ -356,7 +388,8 @@ def build_managed_task( "harness", "harness_version", "model_id", "model_family", "effort", } if (not isinstance(codex_identity, dict) - or set(codex_identity) != required_identity + or set(codex_identity) - (required_identity | {"account_pool_id"}) + or required_identity - set(codex_identity) or codex_identity.get("harness") != "codex" or not all( isinstance(codex_identity[field], str) @@ -384,7 +417,11 @@ def build_managed_task( codex_identity, claude_model=claude_model, claude_effort=claude_effort, + role_bindings={} if role_bindings is None else role_bindings, + pinned_roles=pins, ) + if pins - set(routing["policy"]["roles"]): + raise ContractError("normal pin names an unsupported role") task_class = "managed-review" if workflow == "branch-review" else "managed-fix" task: dict[str, Any] = { "schema_version": 1, @@ -439,19 +476,21 @@ def build_managed_task( task["review"]["focus"] = review_focus.strip() validate_task(task, require_existing_repo=True) task_sha256 = hashlib.sha256(canonical_json(task).encode()).hexdigest() - roles = { - role: { + profiles_by_id = {p["id"]: p for p in routing["profiles"]["profiles"]} + roles = {} + for role in routing["policy"]["roles"]: + alias = NORMAL_ALIASES[role] + override = routing.get("overrides", {}).get(role) + profile = profiles_by_id[(override or routing["profiles"]["bindings"][alias])["profile_id"]] + roles[role] = { "profile_id": profile["id"], "harness": profile["harness"], + "model_id": profile["model_id"], "effort": profile["effort"]["value"], "permission": profile["permission_policy"], + "alias": alias, + "selection_mode": "pinned" if override else "approved_alias" if profile["quality_status"] == "proven" else "bounded_trial", } - for role, profile in ( - ("implementer", next((p for p in routing["profiles"]["profiles"] if p["harness"] == "claude"), None)), - ("reviewer", next((p for p in routing["profiles"]["profiles"] if p["harness"] == "codex"), None)), - ) - if profile is not None - } return task, { "workflow": workflow, "task_sha256": task_sha256, @@ -460,7 +499,8 @@ def build_managed_task( "target_oid": target_oid, "planned_roles": roles, "selection_reason": ( - "one runtime-discovered subscription profile per required role; " + "stable policy aliases; explicit pins are fixed, approved incumbents " + "are retained, otherwise a bounded trial is used; " "different-harness review is mandatory for delivery" ), "scope": task["scope"], diff --git a/test/core/experiment_runtime_fixture.py b/test/core/experiment_runtime_fixture.py index 7fc49e0..664ade2 100644 --- a/test/core/experiment_runtime_fixture.py +++ b/test/core/experiment_runtime_fixture.py @@ -31,6 +31,7 @@ def __init__( self, root: Path, *, with_fallback=False, fail_candidate=False, service=None, repo=None, experiment_id="saved-run-review-pair", candidate_succeeds=True, case_splits=None, profiles=None, workflow="branch-review", + policy=None, task_class=None, ): self.root = root.resolve() self.root.mkdir(parents=True, exist_ok=True) @@ -39,6 +40,7 @@ def __init__( self.candidate_succeeds = candidate_succeeds self.experiment_id = experiment_id self.workflow = workflow + self.task_class = task_class self.role = "implementer" if workflow == "issue-delivery" else "reviewer" self.case_splits = case_splits or [("eval-1", "evaluation"), ("hold-1", "held_out")] self.repo = repo or self.root / "repo" @@ -69,7 +71,7 @@ def __init__( "schema_version": 1, "profiles": list(self.profiles.values()), "bindings": {"review.deep": {"profile_id": self.profiles["control"]["id"], "version": 7}}, } - self.policy = routing_policy() + self.policy = copy.deepcopy(policy) if policy is not None else routing_policy() if workflow == "issue-delivery": for value in self.profiles.values(): value.update(permission_policy="workspace_write", required_tools=["read", "write"]) @@ -143,6 +145,8 @@ def task(self, case_id, arm): "cwd": ".", "timeout_seconds": 10, "required_to_pass": True, }] task["budget"]["wall_seconds"] = 120 + if self.task_class is not None: + task["task_class"] = self.task_class if self.with_fallback: task["budget"].update(max_worker_invocations=2, max_fallbacks_per_step=1) if self.workflow == "issue-delivery": diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 76a34b7..0869d2b 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -123,6 +123,7 @@ def test_review_dry_run_prepares_a_managed_task_without_starting(self): checks=(("python3", "-m", "unittest"),), check_timeout=600, review_mode="standard", review_focus=None, claude_model="sonnet", claude_effort="high", + role_bindings={}, pinned_roles=("reviewer",), ) service.assert_not_called() @@ -183,6 +184,7 @@ def test_fix_starts_managed_delivery_and_reports_run_and_plan(self): check_timeout=600, review_mode="adversarial", review_focus="state transitions", claude_model="claude-sonnet-exact", claude_effort="high", + role_bindings={}, pinned_roles=("reviewer", "implementer"), ) def test_normal_entry_rejects_waiting_for_a_dry_run(self): diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py index 0cfa910..0a7661c 100644 --- a/test/core/test_task_entry.py +++ b/test/core/test_task_entry.py @@ -1,4 +1,7 @@ import io +import copy +import contextlib +import json from pathlib import Path import subprocess import sys @@ -17,6 +20,9 @@ discover_codex_identity, parse_checks, ) +from devsquad.router import load_routing +from devsquad.store import Store, canonical_json +from devsquad import cli class ManagedTaskEntryTest(unittest.TestCase): @@ -95,10 +101,7 @@ def test_branch_review_freezes_exact_commits_and_embedded_routing(self): ["candidate-diff-check", "detected-tests", "user-check-1"], ) self.assertEqual(summary["base_oid"], self.oid) - self.assertEqual( - summary["planned_roles"]["reviewer"]["profile_id"], - "managed-codex-reviewer", - ) + self.assertTrue(summary["planned_roles"]["reviewer"]["profile_id"].startswith("managed-codex-reviewer-")) self.assertEqual(len(summary["task_sha256"]), 64) def test_issue_delivery_is_bounded_to_one_writer_and_required_checks(self): @@ -128,10 +131,7 @@ def test_issue_delivery_is_bounded_to_one_writer_and_required_checks(self): task["checks"][0]["argv"], ["git", "diff", "--check", self.oid, "HEAD", "--"], ) - self.assertEqual( - summary["planned_roles"]["implementer"]["profile_id"], - "managed-claude-implementer", - ) + self.assertTrue(summary["planned_roles"]["implementer"]["profile_id"].startswith("managed-claude-implementer-")) self.assertEqual(summary["planned_roles"]["reviewer"]["harness"], "codex") self.assertIn("different-harness", summary["selection_reason"]) @@ -145,6 +145,107 @@ def test_issue_delivery_is_bounded_to_one_writer_and_required_checks(self): ) self.assertEqual(default_task["scope"]["write_paths"], ["."]) + def test_normal_roles_use_stable_trial_aliases_not_concrete_defaults(self): + task, summary = build_managed_task( + workflow="issue-delivery", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Bounded routing proof.", + codex_identity=self.codex, + ) + self.assertEqual(task["routing"]["policy"]["roles"], { + "implementer": [{"kind": "alias", "id": "implement.balanced"}], + "reviewer": [{"kind": "alias", "id": "review.deep"}], + }) + self.assertIn("bounded trial", summary["selection_reason"]) + self.assertTrue(all(p["quality_status"] == "trial" for p in task["routing"]["profiles"]["profiles"])) + + def test_approved_alias_survives_provider_default_and_explicit_pin_is_truthful(self): + original, _ = build_managed_task( + workflow="branch-review", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Routing proof.", + codex_identity=self.codex, + ) + incumbent = copy.deepcopy(original["routing"]["profiles"]["profiles"][0]) + incumbent.update(id="approved-reviewer", quality_status="proven") + bindings = {"reviewer": {"alias": "review.deep", "version": 7, "profile": incumbent}} + changed_default = {**self.codex, "model_id": "gpt-new-default"} + for pinned in (False, True): + with self.subTest(pinned=pinned): + task, summary = build_managed_task( + workflow="branch-review", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Routing proof.", + codex_identity=changed_default, role_bindings=bindings, + pinned_roles=("reviewer",) if pinned else (), + ) + routing = load_routing(task, canonical_json(task["routing"]["profiles"]), canonical_json(task["routing"]["policy"])) + selected = routing["roles"]["reviewer"]["selected"] + self.assertEqual(selected["profile"]["model_id"], "gpt-new-default" if pinned else "gpt-fixture") + self.assertEqual(routing["roles"]["reviewer"]["source"], "override" if pinned else "automatic") + self.assertEqual(summary["planned_roles"]["reviewer"]["selection_mode"], "pinned" if pinned else "approved_alias") + self.assertEqual(task["routing"]["profiles"]["bindings"]["review.deep"]["version"], 7) + + def test_public_promotion_changes_normal_entry_but_not_the_old_run_or_pin(self): + from experiment_runtime_fixture import ExperimentRuntimeFixture + import test_lifecycle as lifecycle_fixtures + + baseline_task, _ = build_managed_task( + workflow="branch-review", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Normal alias proof.", codex_identity=self.codex, + ) + incumbent = copy.deepcopy(baseline_task["routing"]["profiles"]["profiles"][0]) + incumbent.update(id="approved-a", model_id="gpt-a", quality_status="proven") + candidate = {**copy.deepcopy(incumbent), "id": "approved-b", "model_id": "gpt-b"} + service = Service(Path(self.temp.name) / "runtime") + template = lifecycle_fixtures.lifecycle_template(update_mode="reviewed") + template.update( + policy={"id": "managed-normal-entry", "version": 1}, + allowed_harnesses=["codex"], allowed_model_families=["gpt"], + allowed_account_pools=["codex-subscription"], allowed_task_classes=["managed-review"], + ) + store = Store(service.database, service.artifacts) + self.addCleanup(store.close) + store.bootstrap_profile_binding(template, incumbent, version=7) + fixture = ExperimentRuntimeFixture( + Path(self.temp.name) / "alias-pair", service=service, repo=self.repo, + experiment_id="normal-alias-pair", profiles={"control": incumbent, "candidate": candidate}, + policy=baseline_task["routing"]["policy"], task_class="managed-review", + ) + self.addCleanup(fixture.close) + fixture.run_all() + old_task, _ = build_managed_task( + workflow="branch-review", project_dir=self.repo, + base_ref="HEAD", target_ref="HEAD", goal="Old normal run.", codex_identity=self.codex, + role_bindings=service.normal_entry_bindings("branch-review"), + ) + with mock.patch.object(service, "_spawn_daemon", return_value=0): + old = service.start(old_task, "normal-before-promotion", _internal_review_fixture={"verdict": "clean", "summary": "Offline normal review.", "findings": []}) + self.addCleanup(lambda: service.cancel(old["run_id"])) + self.assertEqual(old["state"], "queued", old) + old_snapshot = store.run(old["run_id"])["mutable_snapshot"] + evaluated = service.policy_evaluate(fixture.spec) + helper = lifecycle_fixtures.ProfileLifecycleTest() + helper.candidate = candidate + qualification = helper.qualification(evaluated) + qualification["task_class"] = "managed-review" + store.record_profile_qualification(qualification) + store.change_profile_binding(helper.promotion("normal-promote-b")) + self.assertEqual(store.run(old["run_id"])["mutable_snapshot"], old_snapshot) + self.assertEqual(json.loads(old_snapshot)["routing"]["roles"]["reviewer"]["selected"]["profile_id"], "approved-a") + + for pin in (False, True): + output = io.StringIO() + requested = {**self.codex, "model_id": "gpt-new-default" if pin else "gpt-b"} + argv = ["review", "--base", "HEAD", "--project-dir", str(self.repo), + "--runtime-dir", str(service.runtime), "--dry-run", "--json"] + if pin: + argv.extend(["--model", "gpt-new-default", "--effort", "low"]) + with contextlib.redirect_stdout(output), mock.patch.object(cli, "discover_codex_identity", return_value=requested) as discovery: + self.assertEqual(cli.main(argv), 0) + planned = json.loads(output.getvalue())["data"]["planned_roles"]["reviewer"] + self.assertEqual(planned["model_id"], "gpt-new-default" if pin else "gpt-b") + self.assertEqual(planned["selection_mode"], "pinned" if pin else "approved_alias") + discovery.assert_called_once_with(self.repo.resolve(), requested_model=planned["model_id"], requested_effort="low") + self.assertEqual(service.normal_entry_bindings("branch-review")["reviewer"]["version"], 8) + def test_managed_delivery_runs_offline_through_review_checks_and_handoff(self): task, _ = build_managed_task( workflow="issue-delivery", From e8cdf07e1b2249cdb29e8c1e145d44584fc093cb Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 15:18:17 -0700 Subject: [PATCH 149/197] WIP checkpoint: Bind delivery experiment evidence and defer unsafe schema upgrades (2026-10-01 15:18) --- docs/plans/engineering-team/RESUME.md | 45 ++++++ docs/plans/engineering-team/backlog.json | 4 +- .../evidence/R3-audit-repairs-2026-10-01.json | 31 ++++ .../core/src/devsquad/experiment_evidence.py | 152 ++++++++++++++++-- plugin/core/src/devsquad/review_worker.py | 2 + plugin/core/src/devsquad/store.py | 19 +++ scripts/install-core.sh | 22 +++ test/core/experiment_runtime_fixture.py | 19 ++- .../test_delivery_experiment_integrity.py | 111 +++++++++++++ test/core/test_experiment_eligibility.py | 17 ++ test/core/test_install_core.py | 91 +++++++++++ test/core/test_review_runtime.py | 10 ++ test/core/test_task_entry.py | 19 ++- 13 files changed, 521 insertions(+), 21 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R3-audit-repairs-2026-10-01.json create mode 100644 test/core/test_delivery_experiment_integrity.py diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index f0edbae..6a64585 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -4,6 +4,51 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ## Current position — October 1, 2026 +### Current continuation — R3 audit repairs and safe schema-update deferral + +The two independent findings now have source repairs and red regressions. +Delivery evidence is decoded against the immutable prelaunch task, hash-bound +implementation imports, candidate/patch artifacts and candidate-ready events. +Review imports/checks must equal captured evidence, and independent identity +is revalidated. Revision prompts are reconstructed from recorded handoff +decisions, not unchecked final mutable snapshots. Failed writer/reviewer output +requires the exact terminal failure receipt. Qualification cannot substitute +caller-provided latency/usage ratios for missing saved paired measurements; +finite measurement gates remain blocked when those measurements are unknown. +Trusted Python checks suppress bytecode and redirect caches to their isolated +temporary HOME, without broadening candidate integrity allowances. + +The initial 22-test targeted reader/eligibility/check gate passed in 71.034 +seconds. The next 34-test gate had one test-helper error (an unpaired failed +arm has no paired case verdict), and the subsequent 19-test gate had one +assertion mismatch: the public replay correctly reports stale evidence rather +than exposing the lower-level failed-receipt reason. Both assertions are now +corrected; neither failed run is represented as a pass. The real historical +schema-13 installation test passed: the update defers without switching the +launcher, the old active run cancels, the queued old run resumes/completes, +then the update migrates to schema 15 and preserves readable results/releases. + +The normal-entry promotion fixture now uses an actual fake-native Codex +protocol, frozen adapters and execution fingerprints instead of pretending +fixture output is native. Its assertions pass after repairing cleanup order. +This is an offline public-chain proof, not real-model qualification. + +**Next action:** finish the affected regression gate, checkpoint, then run one +frozen full core integration gate and a bounded independent repair re-review. +Do not repeat the broad audit. Inspect/reconcile the real old ledger before +installing. The actual local installation remains unchanged; Claude/Grok and +updated Gemini/Antigravity live proofs remain pending. R4 catalog/quota, R5 +public trials/outcomes, R6 terminal UX and R7 Council remain separately open. + +The final affected gate passes **37 tests in 37.167 seconds**, including the +historical installed upgrade and fake-native normal-alias promotion proof. +All **227 Bash assertions in 11 files**, generated reference and whitespace +checks pass. A read-only inspection of the real default ledger found schema +13 with **zero nonterminal runs**; no reconciliation/cancellation is needed. +The installation still points at the original release. Checkpoint this slice +before the unchanged-source full integration run. Portable details are in +[audit repair evidence](evidence/R3-audit-repairs-2026-10-01.json). + ### Latest runtime-repair slice — shared eligibility and explicit review R3b.1's unchanged-source gate at `a4a87fd` passed 427 tests (two optional-SDK diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index c0e719a..1cc422b 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -21,8 +21,8 @@ "status": "partial", "artifact": "evidence/R3b2-eligibility-partial-2026-10-01.json", "scope": "Reader full gate passed at a4a87fd; schema-15 explicit revisions and shared lifecycle eligibility implemented with focused public fixtures; no install or provider calls", - "next_action": "441-test schema-15 integration passes at dc4e110 with two optional-SDK skips and zero unraisable exceptions; finish bounded independent R3 audit and R4 routing/catalog/quota", - "limitations": "Independent R3 audit incomplete; production paired-trial controller remains R5; Codex bundled path/version changed to 0.159.2 and needs compatibility update; native Gemini worker and refreshed installed proof unverified" + "next_action": "Independent R3 audit found delivery evidence and unmeasured-ratio gaps; repairs plus bytecode isolation and schema-13 upgrade safety pass 37 affected tests and 227 Bash assertions. Checkpoint, run frozen full integration and bounded repair re-review, then safely refresh and prove requested live workflows.", + "limitations": "Full integration/re-review on audit repairs pending; R4 catalog/quota, R5 public controller/outcomes, R6 UX and R7 Council open. Installed schema 13 has no nonterminal runs but is not yet refreshed. Claude/Grok/updated Gemini live proofs pending." }, "planning_checkpoint": { "recorded_on": "2026-10-01", diff --git a/docs/plans/engineering-team/evidence/R3-audit-repairs-2026-10-01.json b/docs/plans/engineering-team/evidence/R3-audit-repairs-2026-10-01.json new file mode 100644 index 0000000..96204fa --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3-audit-repairs-2026-10-01.json @@ -0,0 +1,31 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-01", + "baseline_revision": "db55d3f", + "status": "source_targeted_verified_full_gate_pending", + "findings": { + "R3-001": "Delivery imports are mandatory for successful attempts; captured evidence is strictly decoded against immutable task, candidate/patch/event fences and recorded revision decisions. Review, check, evaluation and attempt imports must equal captured evidence. Failed workers require the exact hash-verified terminal attempt receipt.", + "R3-002": "Qualification rejects submitted latency/usage ratios when no verified paired saved measurement exists. Unknown remains unknown; finite gates requiring either metric block qualification.", + "check_bytecode": "Trusted checks set invocation-local PYTHONDONTWRITEBYTECODE and redirect PYTHONPYCACHEPREFIX under the isolated temporary HOME; candidate integrity allowances are unchanged." + }, + "red": { + "delivery": "Hash-consistent malformed stdout, missing imports, candidate drift and changed imports were accepted by the baseline reader; negative regressions failed.", + "ratios": "Independent native audit confirmed invented-ratio bypass. Initial finite-gate regressions exposed submitted ratios not being checked against measurements; full integration/re-review pending.", + "bytecode": "1 test failed: importing a candidate module created __pycache__ in the candidate tree." + }, + "verified": [ + {"gate": "reader/eligibility/check targeted", "tests": 22, "seconds": 71.034, "result": "pass"}, + {"gate": "delivery/task-entry/CLI/ratios/bytecode/historical installed upgrade", "tests": 37, "seconds": 37.167, "result": "pass"}, + {"gate": "Bash", "assertions": 227, "files": 11, "result": "pass"}, + {"gate": "generated reference and whitespace", "result": "pass"}, + {"gate": "schema-13 old-package active/recoverable upgrade", "result": "pass", "scope": "Temporary install defers without selector/schema changes; old active run cancels; old queued run resumes and succeeds; subsequent update migrates to schema 15, retains old release and readable result."}, + {"gate": "normal alias promotion", "result": "pass", "scope": "Four fake-native Codex protocol runs supply public paired evaluation. Promotion affects next normal entry; queued old snapshot and explicit pins remain unchanged. Not real-model qualification."} + ], + "failed_intermediate_gates": [ + {"tests": 34, "seconds": 21.036, "errors": 1, "cause": "Test asserted a paired-case verdict for an unpaired failed arm; corrected fixture runs both arms."}, + {"tests": 19, "seconds": 20.453, "failures": 1, "cause": "Public stale-evidence rejection wrapped the lower-level failed-receipt reason; corrected assertion."}, + {"tests": 1, "seconds": 6.649, "errors": 2, "cause": "Test tearDown deleted runtime before registered cleanups; corrected cleanup ordering."} + ], + "installed_state": {"schema": 13, "nonterminal_runs": 0, "refreshed": false}, + "limitations": ["Full core integration and bounded independent repair re-review pending", "No updated-installation Claude/Grok/Gemini live proof", "No verified paired latency/usage metric; required finite metric gates remain blocked", "R4 catalog/quota, R5 public controller/outcomes, R6 terminal UX and R7 Council remain separately open"] +} diff --git a/plugin/core/src/devsquad/experiment_evidence.py b/plugin/core/src/devsquad/experiment_evidence.py index b434719..16b2fcc 100644 --- a/plugin/core/src/devsquad/experiment_evidence.py +++ b/plugin/core/src/devsquad/experiment_evidence.py @@ -20,7 +20,7 @@ validate_arm_chain, validate_assignment, ) from .learning import validate_outcome -from .store import canonical_json +from .store import canonical_json, request_hash def _digest(value: Any) -> str: @@ -108,6 +108,86 @@ def verify_prelaunch_snapshot( raise ContractError("experiment launch changed its controlled input/profile/execution") +def _named_artifact(connection, run_id, name, artifacts): + row = connection.execute("SELECT id FROM artifacts WHERE run_id=? AND name=?", (run_id, name)).fetchone() + if row is None: + raise ContractError(f"experiment missing imported artifact: {name}") + artifact, content = _artifact(connection, row["id"], run_id) + artifacts.append(artifact) + return content + + +def _delivery_revision(connection, run_id, events, claim_version, previous_candidate): + revisions = [event for event in events if event["type"] == "delivery.revision_queued" + and event["run_version"] < claim_version] + if not revisions: + if previous_candidate is not None: + raise ContractError("experiment delivery continuation has no revision fence") + return None + event = revisions[-1] + payload = _object(event["payload"], "delivery revision event") + row = connection.execute( + "SELECT h.*,s.submission_id,s.submission_hash,s.decision_json,s.disposition,s.recorded_run_version " + "FROM handoffs h JOIN handoff_submissions s ON s.handoff_id=h.id " + "WHERE h.run_id=? AND h.id=? AND s.submission_id=? AND s.outcome='recorded'", + (run_id, payload.get("handoff_id"), payload.get("submission_id")), + ).fetchone() + if row is None: + raise ContractError("experiment delivery revision has no recorded disposition") + packet = _object(row["packet_json"], "revision handoff") + decision = _object(row["decision_json"], "revision disposition") + body = {key: value for key, value in decision.items() if key != "submission_hash"} + if (row["packet_sha256"] != _digest(packet) or row["submission_hash"] != request_hash(body) + or decision.get("submission_hash") != row["submission_hash"] + or decision.get("submission_id") != row["submission_id"] + or row["disposition"] != "revise" or decision.get("disposition") != "revise" + or row["recorded_run_version"] >= event["run_version"] + or previous_candidate is None + or packet.get("candidate_sha256") != previous_candidate["candidate_sha256"] + or payload.get("previous_candidate_sha256") != previous_candidate["candidate_sha256"]): + raise ContractError("experiment delivery revision identity/hash is inconsistent") + return { + "handoff_id": row["id"], "sequence": row["sequence"], + "submission_id": row["submission_id"], "submission_hash": row["submission_hash"], + "previous_candidate_sha256": packet["candidate_sha256"], "reason": decision["reason"], + "review": packet["review"], "checks": packet["checks"], "evidence_refs": decision["evidence_refs"], + } + + +def _delivery_candidate(connection, run_id, attempt_id, events, artifacts, snapshot, iteration): + candidates = connection.execute( + "SELECT id FROM artifacts WHERE run_id=? AND name GLOB 'candidate-*.json'", (run_id,), + ).fetchall() + matches = [] + for row in candidates: + artifact, content = _artifact(connection, row["id"], run_id) + candidate = _object(content, "delivery candidate") + if candidate.get("implementation_artifact") == f"implementation-attempt-{attempt_id}.json": + matches.append((artifact, candidate)) + ready = [event for event in events if event["type"] == "delivery.candidate_ready" + and _object(event["payload"], "candidate event").get("attempt_id") == attempt_id] + if len(matches) != 1 or len(ready) != 1: + raise ContractError("experiment delivery implementation has no unique saved candidate fence") + artifact, candidate = matches[0] + fields = ("schema_version", "baseline_oid", "commit_oid", "tree_oid", "patch_sha256", "changed_paths") + if any(key not in candidate for key in fields): + raise ContractError("experiment saved candidate identity is incomplete") + identity = {key: candidate[key] for key in fields} + event = _object(ready[0]["payload"], "candidate event") + if (candidate.get("candidate_sha256") != _digest(identity) + or candidate["baseline_oid"] != snapshot["delivery_workspace"]["baseline_oid"] + or candidate.get("iteration") != iteration + or any(event.get(key) != candidate[key] for key in ("candidate_sha256", "commit_oid", "patch_sha256")) + or artifact["name"] != f"candidate-{iteration}.json" + or candidate.get("patch_artifact") != f"candidate-{iteration}.patch"): + raise ContractError("experiment saved candidate differs from its implementation fence") + artifacts.append(artifact) + patch = _named_artifact(connection, run_id, candidate["patch_artifact"], artifacts) + if hashlib.sha256(patch).hexdigest() != candidate["patch_sha256"] or len(patch) != candidate.get("patch_bytes"): + raise ContractError("experiment candidate patch differs from its saved identity") + return candidate, ready[0]["run_version"] + + def _attempts( connection: sqlite3.Connection, run: sqlite3.Row, frozen: sqlite3.Row, snapshot: dict[str, Any], assignment: dict[str, Any], @@ -158,6 +238,9 @@ def _attempts( history = [] exposed = False incomplete = False + delivery_snapshot = dict(snapshot) + delivery_iterations = [] + candidate_version = None for row in sorted(rows, key=lambda item: claimed[item["id"]]["run_version"]): role = row["role"] route = snapshot["routing"]["roles"].get(role) @@ -215,7 +298,7 @@ def _attempts( incomplete = True # These immutable artifacts retain imported observed identity and # native usage; failure diagnostics are retained in output_metadata. - review_imported = False + imports = {} for prefix in ("review", "implementation", "lead"): artifact_row = connection.execute( "SELECT id FROM artifacts WHERE run_id=? AND name=?", @@ -229,13 +312,60 @@ def _attempts( or canonical_json(evidence.get("selected_profile")) != canonical_json(selected)): raise ContractError("experiment imported execution differs from its frozen attempt profile") artifacts.append(artifact) - review_imported = review_imported or prefix == "review" - if (role == "reviewer" and snapshot["task"]["workflow"] == "branch-review" - and captures): - if review_imported: - from .workflows import validate_branch_review_evidence + imports[prefix] = document + delivery = snapshot["task"]["workflow"] == "issue-delivery" + if captures and (role == "reviewer" or delivery and role == "implementer"): + prefix = "implementation" if role == "implementer" else "review" + if prefix in imports: + from .workflows import ( + validate_branch_review_evidence, validate_implementation_evidence, + require_check_integrity, require_independent_delivery_review, + ) - validate_branch_review_evidence(strict_json(captures["stdout"]), snapshot) + context = snapshot + if delivery and role == "implementer": + context = dict(snapshot) + previous = delivery_iterations[-1]["candidate"] if delivery_iterations else None + revision = _delivery_revision(connection, run["id"], events, claimed[row["id"]]["run_version"], previous) + if revision is not None: + context["revision_request"] = revision + document = validate_implementation_evidence(strict_json(captures["stdout"]), context) + if canonical_json(imports[prefix]) != canonical_json(document): + raise ContractError("experiment imported implementation differs from captured evidence") + candidate, candidate_version = _delivery_candidate( + connection, run["id"], row["id"], events, artifacts, snapshot, len(delivery_iterations) + 1, + ) + if candidate_version <= claimed[row["id"]]["run_version"]: + raise ContractError("experiment candidate precedes its implementation reservation") + iteration = {"candidate": candidate, "implementation": document} + if revision is not None: + iteration["revision_request"] = revision + delivery_iterations.append(iteration) + delivery_snapshot = {**snapshot, "delivery_iterations": delivery_iterations, "workspace": { + "candidate_sha256": candidate["candidate_sha256"], "base_oid": candidate["baseline_oid"], + "target_oid": candidate["commit_oid"], + }} + if "pending_review_fixture" in snapshot: + delivery_snapshot["internal_review_fixture"] = snapshot["pending_review_fixture"] + else: + if delivery: + if candidate_version is None or candidate_version >= claimed[row["id"]]["run_version"]: + raise ContractError("experiment delivery reviewer has no preceding implementation candidate") + context = delivery_snapshot + document = validate_branch_review_evidence(strict_json(captures["stdout"]), context) + if canonical_json(imports[prefix]) != canonical_json(document["attempt"]): + raise ContractError("experiment imported review attempt differs from captured evidence") + for name, expected in ( + (f"review-{row['id']}.json", document["review"]), + (f"checks-{row['id']}.json", {"schema_version": 1, "candidate_sha256": document["candidate_sha256"], + "target_oid": document["target_oid"], "results": document["checks"]}), + (f"evaluation-{row['id']}.json", document["evaluation"]), + ): + imported = _object(_named_artifact(connection, run["id"], name, artifacts), "review import") + if canonical_json(imported) != canonical_json(expected): + raise ContractError("experiment imported review/check/evaluation differs from captured evidence") + require_check_integrity(document["checks"]) + require_independent_delivery_review(context, document) else: # Terminal failed/cancelled workers do not publish successful # review evidence. Bind their opaque output to the hashed @@ -245,12 +375,12 @@ def _attempts( (run["id"],), ).fetchone() if receipt_row is None: - raise ContractError("experiment reviewer has no imported review or failure receipt") + raise ContractError(f"experiment {role} has no imported evidence or failure receipt") artifact, content = _artifact(connection, receipt_row["id"], run["id"]) receipt = _object(content, "terminal failure receipt") receipt_attempts = receipt.get("attempts") if not isinstance(receipt_attempts, list): - raise ContractError("experiment failed reviewer receipt attempts are invalid") + raise ContractError(f"experiment failed {role} receipt attempts are invalid") projections = [item for item in receipt_attempts if isinstance(item, dict) and item.get("id") == row["id"]] if (receipt.get("run_id") != run["id"] @@ -259,7 +389,7 @@ def _attempts( or projections[0].get("status") not in {"failed", "cancelled"} or projections[0].get("role") != role or canonical_json(projections[0].get("selected_profile")) != canonical_json(selected)): - raise ContractError("experiment failed reviewer receipt is inconsistent") + raise ContractError(f"experiment failed {role} receipt is inconsistent") artifacts.append(artifact) history.append({ **{key: row[key] for key in ( diff --git a/plugin/core/src/devsquad/review_worker.py b/plugin/core/src/devsquad/review_worker.py index 6911f85..feaee50 100644 --- a/plugin/core/src/devsquad/review_worker.py +++ b/plugin/core/src/devsquad/review_worker.py @@ -69,6 +69,8 @@ def _run_check( with tempfile.TemporaryDirectory(prefix="devsquad-check-home-") as check_home: environment = os.environ.copy() environment["HOME"] = check_home + environment["PYTHONDONTWRITEBYTECODE"] = "1" + environment["PYTHONPYCACHEPREFIX"] = str(Path(check_home) / "python-cache") try: process = subprocess.Popen( check["argv"], diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 5d638a6..682e4cf 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -174,6 +174,18 @@ def migrate(self) -> None: current = self.connection.execute("SELECT COALESCE(MAX(version), 0) FROM schema_migrations").fetchone()[0] if table else 0 if current > SUPPORTED_SCHEMA_VERSION: raise SchemaVersionError(f"database schema {current} is newer than supported {SUPPORTED_SCHEMA_VERSION}") + if 0 < current < SUPPORTED_SCHEMA_VERSION: + runs_table = self.connection.execute("SELECT 1 FROM sqlite_master WHERE type='table' AND name='runs'").fetchone() + if runs_table is not None: + pending = self.connection.execute( + "SELECT id,state,phase FROM runs WHERE state NOT IN ('succeeded','failed','cancelled') ORDER BY created_at", + ).fetchall() + if pending: + identifiers = ", ".join(row["id"] for row in pending) + raise SchemaVersionError( + f"schema upgrade {current} → {SUPPORTED_SCHEMA_VERSION} deferred: active/recoverable runs {identifiers}; " + "finish or cancel them using the previous installed release, then retry the update" + ) while current < SUPPORTED_SCHEMA_VERSION: next_version = current + 1 candidates = [entry for entry in files("devsquad.migrations").iterdir() if entry.name.startswith(f"{next_version:03d}_") and entry.name.endswith(".sql")] @@ -2154,6 +2166,13 @@ def _qualification_evidence( or record["measured"]["held_out_pairs"] != metrics["held_out"]["available_pairs"] or record["measured"]["critical_defects"] != escaped): raise ContractError("qualification measurements do not match saved evaluation") + # Saved evaluations currently contain no verified paired whole-run + # latency or usage measurement. UTC attempt timestamps are not + # monotonic latency, and partial native token reports are not a + # complete paired usage measure. Preserve unknown rather than + # letting caller-supplied ratios grant lifecycle authority. + if any(record["measured"][field] is not None for field in ("latency_ratio", "usage_ratio")): + raise ContractError("qualification measurements contain unmeasured latency/usage ratios") if evaluation["verdict"] != "promotion_proposal": failures.append("experiment_did_not_propose_promotion") failures = sorted(set(failures)) diff --git a/scripts/install-core.sh b/scripts/install-core.sh index 53643b9..d5bb63f 100755 --- a/scripts/install-core.sh +++ b/scripts/install-core.sh @@ -391,6 +391,28 @@ PY RELEASE_CREATED=1 fi +# Migrate the explicitly scoped default ledger before switching the launcher. +# Store fences migrations transactionally against active/recoverable old runs. +# Custom runtimes receive the same guard on their first new-release operation. +UPGRADE_RUNTIME="${DEVSQUAD_RUNTIME_DIR:-${INSTALL_ROOT}/runtime}" +case "$UPGRADE_RUNTIME" in /*) ;; *) fail "runtime directory must be absolute" ;; esac +if [ -f "$UPGRADE_RUNTIME/state.sqlite3" ]; then + "$RELEASE_DIR/venv/bin/python" -P - "$UPGRADE_RUNTIME" <<'PY' || fail "runtime upgrade deferred; previous current selector and launcher are unchanged" +from pathlib import Path +import sys +from devsquad.store import Store + +runtime = Path(sys.argv[1]) +try: + store = Store(runtime / "state.sqlite3", runtime / "artifacts") +except Exception as exc: + print(str(exc), file=sys.stderr) + raise SystemExit(1) +else: + store.close() +PY +fi + CURRENT_CHANGED=0 CURRENT_TARGET="releases/$RELEASE_ID" if [ -L "$INSTALL_ROOT/current" ] && [ "$(readlink "$INSTALL_ROOT/current")" = "$CURRENT_TARGET" ]; then diff --git a/test/core/experiment_runtime_fixture.py b/test/core/experiment_runtime_fixture.py index 664ade2..21b0054 100644 --- a/test/core/experiment_runtime_fixture.py +++ b/test/core/experiment_runtime_fixture.py @@ -20,7 +20,8 @@ from test_learning import experimental_final from test_lifecycle import profile, review_task, routing_policy -from devsquad.experiment_provenance import assignment_for, paired_input_identity +from devsquad.experiment_provenance import assignment_for, paired_input_identity, selected_execution_fingerprint +from devsquad.codex_review_worker import freeze_codex_reviewer from devsquad.router import load_routing from devsquad.service import Service from devsquad.store import Store, canonical_json, git_common_dir, request_hash @@ -31,7 +32,7 @@ def __init__( self, root: Path, *, with_fallback=False, fail_candidate=False, service=None, repo=None, experiment_id="saved-run-review-pair", candidate_succeeds=True, case_splits=None, profiles=None, workflow="branch-review", - policy=None, task_class=None, + policy=None, task_class=None, native_review=False, ): self.root = root.resolve() self.root.mkdir(parents=True, exist_ok=True) @@ -41,6 +42,7 @@ def __init__( self.experiment_id = experiment_id self.workflow = workflow self.task_class = task_class + self.native_review = native_review self.role = "implementer" if workflow == "issue-delivery" else "reviewer" self.case_splits = case_splits or [("eval-1", "evaluation"), ("hold-1", "held_out")] self.repo = repo or self.root / "repo" @@ -86,6 +88,11 @@ def __init__( fallback = profile("profile-fallback", "model-fallback") self.registry["profiles"].append(fallback) self.policy["roles"]["reviewer"] = [{"kind": "profile", "id": fallback["id"]}] + self.review_adapters = {} + if native_review: + for value in self.profiles.values(): + selected = {"profile_id": value["id"], "profile": value, "profile_sha256": digest(value)} + self.review_adapters[value["id"]] = freeze_codex_reviewer(selected) self.package_path, self.package_digest = self.service._freeze_package() cases = [] for case_id, split in self.case_splits: @@ -109,7 +116,8 @@ def __init__( "kind": "profile_binding", "alias": "review.deep", "role": self.role, **{f"{arm}_profile_id": value["id"] for arm, value in self.profiles.items()}, **{f"{arm}_profile_sha256": digest(value) for arm, value in self.profiles.items()}, - **{f"{arm}_execution_sha256": execution_digest(value) for arm, value in self.profiles.items()}, + **{f"{arm}_execution_sha256": selected_execution_fingerprint(self.declaration_snapshot(self.case_splits[0][0], arm), role=self.role) + for arm in self.profiles}, }, "cases": cases, "gate": { @@ -177,6 +185,7 @@ def declaration_snapshot(self, case_id, arm): "workspace": {**identity, "candidate_sha256": digest(identity)}, "configs": {"policy_file": {"sha256": digest(self.policy)}}, "routing": load_routing(task, canonical_json(self.registry), canonical_json(self.policy)), + **({"review_adapters": copy.deepcopy(self.review_adapters)} if self.native_review else {}), } def store(self): @@ -212,10 +221,12 @@ def predeclared_assignment(*args, **kwargs): fixture_args["_internal_implementation_fixture"] = { "writes": [{"path": "README", "content": f"fixed {case_id}\n"}], "delay_seconds": 0, + **({"fail_profile_ids": [self.profiles["candidate"]["id"]]} if self.fail_candidate else {}), } + if not self.native_review: + fixture_args["_internal_review_fixture"] = {"verdict": "clean", "summary": "Fixture review of the frozen candidate.", "findings": []} started = self.service.start( self.task(case_id, arm), self.outcome_id(case_id, arm), - _internal_review_fixture={"verdict": "clean", "summary": "Fixture review of the frozen candidate.", "findings": []}, **fixture_args, ) run_id = started["run_id"] diff --git a/test/core/test_delivery_experiment_integrity.py b/test/core/test_delivery_experiment_integrity.py new file mode 100644 index 0000000..5fff207 --- /dev/null +++ b/test/core/test_delivery_experiment_integrity.py @@ -0,0 +1,111 @@ +"""Delivery exposure requires semantically valid, durably imported evidence.""" + +from contextlib import contextmanager +import hashlib +import json +from pathlib import Path +import tempfile +import unittest + +from experiment_runtime_fixture import ExperimentRuntimeFixture +from devsquad.contracts import ContractError +from devsquad.store import canonical_json + + +class DeliveryExperimentIntegrityTest(unittest.TestCase): + def setUp(self): + temporary = tempfile.TemporaryDirectory(prefix="devsquad-delivery-evidence-") + self.addCleanup(temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(temporary.name), workflow="issue-delivery") + self.addCleanup(self.fixture.close) + self.run_id = self.fixture.run_arm("eval-1", "candidate") + self.store = self.fixture.store() + self.addCleanup(self.store.close) + + @contextmanager + def changed_artifact(self, row, content, *, attempt=None): + path = Path(row["path"]) + original = path.read_bytes() + digest = hashlib.sha256(content).hexdigest() + path.write_bytes(content) + self.store.connection.execute("UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (digest, len(content), row["id"])) + if attempt is not None: + metadata = json.loads(attempt["output_metadata"]) + metadata["stdout"].update(captured_sha256=digest, captured_bytes=len(content), + full_sha256=digest, total_bytes=len(content)) + self.store.connection.execute("UPDATE attempts SET output_metadata=? WHERE id=?", + (canonical_json(metadata), attempt["id"])) + try: + yield + finally: + path.write_bytes(original) + self.store.connection.execute("UPDATE artifacts SET sha256=?,byte_size=? WHERE id=?", + (row["sha256"], row["byte_size"], row["id"])) + if attempt is not None: + self.store.connection.execute("UPDATE attempts SET output_metadata=? WHERE id=?", + (attempt["output_metadata"], attempt["id"])) + + def rejected(self): + with self.assertRaises(ContractError): + self.fixture.service.policy_evaluate(self.fixture.spec) + self.assertEqual(self.store.connection.execute("SELECT COUNT(*) FROM experiments").fetchone()[0], 0) + + def test_hash_consistent_malformed_delivery_streams_cannot_count_as_exposure(self): + for role in ("implementer", "reviewer"): + attempt = dict(self.store.connection.execute( + "SELECT * FROM attempts WHERE run_id=? AND role=?", (self.run_id, role)).fetchone()) + row = dict(self.store.connection.execute( + "SELECT * FROM artifacts WHERE id=?", (attempt["stdout_artifact_id"],)).fetchone()) + with self.subTest(role=role), self.changed_artifact(row, b'{"not":"worker evidence"}', attempt=attempt): + self.rejected() + + def test_successful_delivery_requires_both_imports(self): + for prefix in ("implementation", "review"): + row = self.store.connection.execute( + "SELECT id,name FROM artifacts WHERE run_id=? AND name LIKE ?", + (self.run_id, f"{prefix}-attempt-%.json")).fetchone() + with self.subTest(prefix=prefix): + self.store.connection.execute("UPDATE artifacts SET name='hidden-import.json' WHERE id=?", (row["id"],)) + try: + self.rejected() + finally: + self.store.connection.execute("UPDATE artifacts SET name=? WHERE id=?", (row["name"], row["id"])) + + def test_saved_candidate_and_imported_usage_are_bound_to_the_stream(self): + for name in ("candidate-1.json", "implementation-attempt-%", "review-attempt-%", "checks-%"): + row = dict(self.store.connection.execute( + "SELECT * FROM artifacts WHERE run_id=? AND name LIKE ?", (self.run_id, name)).fetchone()) + changed = json.loads(Path(row["path"]).read_bytes()) + if name.startswith("candidate"): + changed["commit_oid"] = "0" * 40 + elif name.startswith("checks"): + changed["results"][0]["target_oid"] = "0" * 40 + else: + attempt = changed.get("attempt", changed) + attempt["usage"] = {"input_tokens": 1, "output_tokens": 1, + "total_tokens": 2, "source": "native_reported"} + with self.subTest(name=name), self.changed_artifact(row, canonical_json(changed).encode()): + self.rejected() + + def test_real_failed_writer_requires_an_exact_terminal_receipt(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "failed-writer", workflow="issue-delivery", fail_candidate=True) + self.addCleanup(fixture.close) + fixture.run_all() + run_id = fixture.runs[("eval-1", "candidate")] + store = fixture.store() + self.addCleanup(store.close) + self.assertEqual(fixture.service.status(run_id)["state"], "failed") + evaluated = fixture.service.policy_evaluate(fixture.spec) + self.assertEqual(evaluated["evaluation"]["cases"][0]["candidate_verdict"], "failed") + row = dict(store.connection.execute("SELECT * FROM artifacts WHERE run_id=? AND name='result-receipt.json'", (run_id,)).fetchone()) + content = json.loads(Path(row["path"]).read_bytes()) + content["attempts"][0]["status"] = "succeeded" + original_store = self.store + self.store = store + try: + with self.changed_artifact(row, canonical_json(content).encode()): + with self.assertRaisesRegex(ContractError, "failed implementer receipt|stale"): + fixture.service.policy_evaluate(fixture.spec) + finally: + self.store = original_store diff --git a/test/core/test_experiment_eligibility.py b/test/core/test_experiment_eligibility.py index 8c9f6dc..6a7b5cf 100644 --- a/test/core/test_experiment_eligibility.py +++ b/test/core/test_experiment_eligibility.py @@ -51,6 +51,23 @@ def correct(self, fixture=None, *, arm="candidate"): ) fixture.service.outcome_add(fixture.runs[("hold-1", arm)], correction) + def test_unmeasured_ratios_cannot_bypass_finite_qualification_gates(self): + template = copy.deepcopy(self.template) + template["template_id"] = "finite-measurement-template" + template["gate"].update(max_latency_ratio=2.0, max_usage_ratio=2.0) + self.store.register_profile_template(template) + qualification = copy.deepcopy(self.qualification) + qualification["template_id"] = template["template_id"] + for fields in (("latency_ratio",), ("usage_ratio",), ("latency_ratio", "usage_ratio")): + with self.subTest(invented=fields): + changed = copy.deepcopy(qualification) + for field in fields: + changed["measured"][field] = 0.0 + with self.assertRaisesRegex(ContractError, "measurements|unmeasured"): + self.store.record_profile_qualification(changed) + with self.assertRaisesRegex(ContractError, "latency_ratio_missing.*usage_ratio_missing"): + self.store.record_profile_qualification(qualification) + def next_fixture(self, experiment_id, *, profiles=None, candidate_succeeds=False): fixture = ExperimentRuntimeFixture( self.fixture.root / experiment_id, service=self.fixture.service, repo=self.fixture.repo, diff --git a/test/core/test_install_core.py b/test/core/test_install_core.py index a640371..3cf66a8 100644 --- a/test/core/test_install_core.py +++ b/test/core/test_install_core.py @@ -1,4 +1,5 @@ import json +import io from contextlib import closing import os from pathlib import Path @@ -8,6 +9,7 @@ import subprocess import sys import tempfile +import tarfile import time import unittest import zipfile @@ -435,6 +437,95 @@ def test_update_does_not_break_an_active_release_pinned_run(self): self.assertTrue(Path(package_path).is_dir()) self.assertEqual(len(package_digest), 64) + def test_schema_13_update_defers_for_active_and_recoverable_old_runs(self): + """Exercise the historical package, not a same-schema version bump.""" + old_source = self.root / "schema13-core" + old_source.mkdir() + archived = subprocess.run(["git", "archive", "f4fa657:plugin/core"], cwd=ROOT, + check=True, capture_output=True).stdout + with tarfile.open(fileobj=io.BytesIO(archived)) as archive: + archive.extractall(old_source, filter="data") + first = self.install(old_source) + old_release = Path(first["current_target"]) + old_python = old_release / "venv/bin/python" + runtime = self.install_root / "runtime" + repo = self.root / "schema13-repo" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + subprocess.run(["git", "-C", str(repo), "config", "user.name", "Test"], check=True) + subprocess.run(["git", "-C", str(repo), "config", "user.email", "test@example.invalid"], check=True) + (repo / "README").write_text("baseline\n") + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(repo), "commit", "-qm", "baseline"], check=True) + profiles, policy = branch_review_routing_documents() + task = json.loads((ROOT / "docs/plans/engineering-team/examples/branch-review.json").read_text()) + task.update(project={"repo_path": str(repo), "base_ref": "HEAD", "target_ref": "HEAD"}, + scope={"read_paths": ["README"], "write_paths": []}, checks=[], + routing={"profiles": json.loads(profiles), "policy": json.loads(policy)}) + task_path = self.root / "schema13-task.json" + task_path.write_text(json.dumps(task)) + script = """ +import json,sys +from pathlib import Path +from unittest.mock import patch +from devsquad.service import Service +s=Service(Path(sys.argv[2])) +task=json.loads(Path(sys.argv[1]).read_text()) +live=s.start(task,'schema13-active',_internal_fake_delay=30) +with patch.object(s,'_spawn_daemon',return_value=0): + queued=s.start(task,'schema13-recoverable',_internal_fake_delay=0.1) +print(json.dumps([live,queued])) +""" + process = subprocess.run([str(old_python), "-P", "-c", script, str(task_path), str(runtime)], + check=True, capture_output=True, text=True, env=self.environment, cwd=self.root) + live, queued = json.loads(process.stdout) + launcher = self.bin_dir / "squad" + try: + deadline = time.monotonic() + 8 + while time.monotonic() < deadline: + status = self.cli_json(launcher, "status", live["run_id"], "--runtime-dir", str(runtime)) + if status["data"]["state"] == "running": + break + time.sleep(0.05) + self.assertEqual(status["data"]["state"], "running", status) + result = subprocess.run(["/bin/bash", str(INSTALLER), "--json"], capture_output=True, + text=True, env=self.environment, cwd=ROOT) + self.assertNotEqual(result.returncode, 0) + self.assertIn("active/recoverable", result.stderr) + self.assertEqual((self.install_root / "current").resolve(), old_release) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) + self.cli_json(launcher, "cancel", live["run_id"], "--runtime-dir", str(runtime)) + deadline = time.monotonic() + 8 + while time.monotonic() < deadline: + state = self.cli_json(launcher, "status", live["run_id"], "--runtime-dir", str(runtime))["data"]["state"] + if state == "cancelled": + break + time.sleep(0.05) + self.assertEqual(state, "cancelled") + # A queued old-package run remains recoverable after the deferral. + self.cli_json(launcher, "resume", queued["run_id"], "--runtime-dir", str(runtime)) + deadline = time.monotonic() + 8 + while time.monotonic() < deadline: + state = self.cli_json(launcher, "status", queued["run_id"], "--runtime-dir", str(runtime))["data"]["state"] + if state == "succeeded": + break + time.sleep(0.05) + self.assertEqual(state, "succeeded") + updated = self.install() + self.assertNotEqual(updated["current_target"], str(old_release)) + self.assertTrue(old_release.is_dir()) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) + self.assertTrue(self.cli_json(launcher, "result", queued["run_id"], "--runtime-dir", str(runtime))["data"]["ready"]) + finally: + # Use the old package explicitly even if a later assertion fails. + cleanup_script = "from pathlib import Path; import sys; from devsquad.service import Service; s=Service(Path(sys.argv[1])); [s.cancel(r) for r in sys.argv[2:]]" + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + schema = connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] + if schema == 13: + subprocess.run([str(old_python), "-P", "-c", cleanup_script, str(runtime), live["run_id"], queued["run_id"]], + capture_output=True, env=self.environment, cwd=self.root) + def cli_json(self, launcher, *arguments): result = subprocess.run( [str(launcher), *arguments, "--json"], diff --git a/test/core/test_review_runtime.py b/test/core/test_review_runtime.py index ce25afe..d532ab8 100644 --- a/test/core/test_review_runtime.py +++ b/test/core/test_review_runtime.py @@ -175,6 +175,16 @@ def test_each_trusted_check_gets_a_fresh_isolated_home(self): self.assertTrue(all(not Path(home).exists() for home in homes)) self.assertEqual((inherited_home / "credential-marker").read_text(), "private\n") + def test_trusted_python_check_does_not_write_bytecode_into_candidate(self): + (self.repo / "checked_module.py").write_text("VALUE = 7\n") + with patch.dict(os.environ, {"PYTHONDONTWRITEBYTECODE": "0"}): + result = _run_check({ + "id": "import-candidate", "argv": [sys.executable, "-c", "import checked_module; assert checked_module.VALUE == 7"], + "cwd": ".", "timeout_seconds": 10, "required_to_pass": True, + }, self.repo.resolve(), "candidate-sha256", self.target) + self.assertEqual(result["status"], "passed") + self.assertFalse((self.repo / "__pycache__").exists()) + def configure_fixture_headless(self): profiles = json.loads((self.repo / "devsquad/profiles.json").read_text()) lead = dict(profiles["profiles"][0]) diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py index 0a7661c..31ee3cf 100644 --- a/test/core/test_task_entry.py +++ b/test/core/test_task_entry.py @@ -2,6 +2,7 @@ import copy import contextlib import json +import os from pathlib import Path import subprocess import sys @@ -28,6 +29,7 @@ class ManagedTaskEntryTest(unittest.TestCase): def setUp(self): self.temp = tempfile.TemporaryDirectory(prefix="devsquad-task-entry-") + self.addCleanup(self.temp.cleanup) self.repo = Path(self.temp.name) / "project" self.repo.mkdir() subprocess.run( @@ -65,9 +67,6 @@ def setUp(self): "effort": "low", } - def tearDown(self): - self.temp.cleanup() - def test_branch_review_freezes_exact_commits_and_embedded_routing(self): task, summary = build_managed_task( workflow="branch-review", @@ -187,6 +186,17 @@ def test_public_promotion_changes_normal_entry_but_not_the_old_run_or_pin(self): from experiment_runtime_fixture import ExperimentRuntimeFixture import test_lifecycle as lifecycle_fixtures + fake_bin = Path(self.temp.name) / "fake-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to(ROOT / "test/core/fakes/codex_review_cli.py") + fake_home = Path(self.temp.name) / "fake-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) + environment = mock.patch.dict(os.environ, {"PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", "CODEX_HOME": str(fake_home)}) + environment.start() + self.addCleanup(environment.stop) + baseline_task, _ = build_managed_task( workflow="branch-review", project_dir=self.repo, base_ref="HEAD", target_ref="HEAD", goal="Normal alias proof.", codex_identity=self.codex, @@ -207,7 +217,7 @@ def test_public_promotion_changes_normal_entry_but_not_the_old_run_or_pin(self): fixture = ExperimentRuntimeFixture( Path(self.temp.name) / "alias-pair", service=service, repo=self.repo, experiment_id="normal-alias-pair", profiles={"control": incumbent, "candidate": candidate}, - policy=baseline_task["routing"]["policy"], task_class="managed-review", + policy=baseline_task["routing"]["policy"], task_class="managed-review", native_review=True, ) self.addCleanup(fixture.close) fixture.run_all() @@ -226,6 +236,7 @@ def test_public_promotion_changes_normal_entry_but_not_the_old_run_or_pin(self): helper.candidate = candidate qualification = helper.qualification(evaluated) qualification["task_class"] = "managed-review" + qualification["budget"].update(max_worker_invocations=4, worker_invocations=4) store.record_profile_qualification(qualification) store.change_profile_binding(helper.promotion("normal-promote-b")) self.assertEqual(store.run(old["run_id"])["mutable_snapshot"], old_snapshot) From 6874de7074f1cbf6a67456429ebc8dab7c25073b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 15:36:25 -0700 Subject: [PATCH 150/197] WIP checkpoint: Fence lazy schema upgrades and retain activation failure recovery (2026-10-01 15:36) --- CONTRIBUTING.md | 2 +- docs/RUNTIME-GUIDE.md | 12 +++ docs/plans/engineering-team/RESUME.md | 42 ++++++++++ .../R8-upgrade-review-2026-10-01.json | 22 +++++ .../core/src/devsquad/release_activation.py | 29 +++++++ plugin/core/src/devsquad/store.py | 20 +++++ scripts/install-core.sh | 34 ++++---- scripts/run-core-tests.py | 39 +++++++++ test/core/test_capacity.py | 19 ++++- test/core/test_install_core.py | 81 ++++++++++++++++++- 10 files changed, 277 insertions(+), 23 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json create mode 100644 plugin/core/src/devsquad/release_activation.py create mode 100644 scripts/run-core-tests.py diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 2cb4c0f..b46285a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -7,7 +7,7 @@ git clone https://github.com/joshidikshant/devsquad.git cd devsquad bash test/run.sh # no network, no real CLIs required — should be all green PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src \ - python3 -m unittest discover -s test/core + python3 scripts/run-core-tests.py python3 scripts/generate-core-reference.py --check ``` diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index 608f9ca..4bd28e7 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -47,6 +47,18 @@ inherited, malformed or ambiguous MCP registrations instead of guessing which one to replace. An upgrade retains every previous release, so a process that started before the selector changed can finish against its frozen package. +A schema-changing update defers while the ledger contains active or +recoverable runs. Finish, resume or cancel them with the previous release, +then retry the same installer command. Activation checks the ledger under +its lock and never advances the schema before selecting the new release. +The first new-release ledger operation performs the guarded migration; +already-open old clients cannot write after that migration commits. An +interruption before selector replacement leaves the old schema usable; an +interruption after replacement leaves migration safely retryable. Old +releases and saved receipts remain present. Custom runtime directories get +the same migration guard on first access; use `DEVSQUAD_RUNTIME_DIR` for the +installer's explicitly scoped ledger check. + Antigravity's non-interactive print mode also enforces project permissions. For unattended read-only status checks, add this exact grant to the DevSquad project's Permissions list in Antigravity: diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 6a64585..bd3f5d0 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -49,6 +49,48 @@ The installation still points at the original release. Checkpoint this slice before the unchanged-source full integration run. Portable details are in [audit repair evidence](evidence/R3-audit-repairs-2026-10-01.json). +### Upgrade re-review and current full-gate rerun + +The bounded real Codex repair review against `db55d3f` → `e8cdf07` completed +in private runtime `r3-bounded-repair-review-20261001`, run +`fcc030af-26ec-423c-b462-b6fc0f8b55d1`. It found one high-severity upgrade +activation/admission race, not a separate R3 evidence/ratio finding. The +normal Bash check passed with **verified candidate integrity and no changed +inputs**, confirming the bytecode repair. The host rejected this review; +terminal state is failed, version 22. Native usage was 390,975 input / 4,422 +output tokens (395,397 total), one worker invocation, unknown internal request +count. Re-review only the additional upgrade repairs, not the broad source. + +Activation now checks/swaps under the ledger lock **without migrating**. The +new selected release performs lazy transactional migration; database guards +reject late writes by already-open old clients. Injected activation failure +leaves the old selector/schema untouched and releases the lock. The historical +schema-13 active/queued continuation/cancel test and pre-opened old-client +admission test pass. The final 12-test upgrade/capacity gate passes in 9.109 +seconds. The initial helper extraction run had two errors because the historic +package lacks the new helper; the installer now uses its own helper with the +selected package's explicit supported schema version. The old schema-8 SQL +backfill is still tested separately, while public upgrade correctly defers. + +The full integration launch from stdin was **invalid and interrupted (exit +130)**: spawned subprocess tests cannot reopen ``. It is not a passing +gate or a proven runtime failure. A tracked spawn-safe runner now records +UTC/monotonic timing, forced collection and unraisable diagnostics. Use +`PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning +PYTHONPATH=plugin/core/src:test/core python3 scripts/run-core-tests.py`. +Run one unchanged-source full gate after this checkpoint, then bounded upgrade +re-review. The real installation is still unchanged and all requested live +installed gates are pending. Do not repeat the spent Jev pilot or use paid APIs. + +The subsequent evidence/process regression gate passes **32 tests in 137.768 +seconds**, including real spawned-process recovery tests from a valid module +entrypoint. All 227 Bash assertions and generated-reference/whitespace checks +pass again. No process from the invalid stdin gate remains. The next full +gate must use the tracked runner, not stdin. The Antigravity IDE computer-use +surface was denied by the tool's app permission gate; do not bypass it. Recheck +the existing supported CLI/MCP surface after installation, and distinguish any +unavailable IDE UI proof from a CLI receipt. + ### Latest runtime-repair slice — shared eligibility and explicit review R3b.1's unchanged-source gate at `a4a87fd` passed 427 tests (two optional-SDK diff --git a/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json new file mode 100644 index 0000000..64298b5 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json @@ -0,0 +1,22 @@ +{ + "schema_version": 1, + "reviewed_revision": "e8cdf07", + "status": "finding_repaired_targeted_full_and_re_review_pending", + "run_id": "fcc030af-26ec-423c-b462-b6fc0f8b55d1", + "runtime": "/Users/Dikshant/.devsquad/private-probes/r3-bounded-repair-review-20261001", + "review_sha256": "13b77ee0812c9beefda98df697af4189c132c7e311df995c7b78e9f6282b8644", + "review_attempt_sha256": "5517920d44aa8eaa66bffbf1d147fe8e64bc8acc16be3d62f2c8c38ddcd6bb74", + "checks_sha256": "1e94a349e0bc098ce23f3b24332ef36aed099c23eb713463f07e1965e774805d", + "reviewer": {"harness": "codex", "version": "0.159.2", "model": "gpt-6.1-sol", "effort": "high", "permission": "read_only", "verification": "verified"}, + "finding": {"id": "upgrade-activation-window", "severity": "high", "summary": "Migration committed while the old release selector remained active; pre-opened old clients could admit work after the pending-run check."}, + "host_disposition": "reject", + "terminal": {"state": "failed", "version": 22}, + "check": {"command": "bash test/run.sh", "returncode": 0, "assertions": 227, "integrity": "verified", "changed_inputs": []}, + "usage": {"input_tokens": 390975, "output_tokens": 4422, "total_tokens": 395397, "source": "native_reported", "worker_invocations": 1, "native_model_requests": null}, + "repair": "Readiness/selector swap under one ledger lock without advancing schema. New-release lazy migration transactionally installs connection-version write guards. Old active/recoverable runs defer; late preopened old clients fail before writing. Previous releases and receipts retained.", + "targeted": {"tests": 12, "seconds": 9.109, "result": "pass"}, + "evidence_process_regressions": {"tests": 32, "seconds": 137.768, "result": "pass"}, + "bash": {"assertions": 227, "files": 11, "result": "pass"}, + "full_gate": "Original stdin launch interrupted as invalid: multiprocessing cannot reopen ; repeat using tracked spawn-safe runner.", + "installed_refresh": false +} diff --git a/plugin/core/src/devsquad/release_activation.py b/plugin/core/src/devsquad/release_activation.py new file mode 100644 index 0000000..2d70b41 --- /dev/null +++ b/plugin/core/src/devsquad/release_activation.py @@ -0,0 +1,29 @@ +"""Select a prepared release without advancing an old ledger's schema.""" + +import os +from pathlib import Path +import sqlite3 + +def activate_release(temporary: Path, selector: Path, runtime: Path, *, supported_schema_version: int) -> None: + connection = None + try: + database = runtime / "state.sqlite3" + if database.exists(): + connection = sqlite3.connect(database, isolation_level=None, timeout=10) + connection.execute("BEGIN EXCLUSIVE") + version = connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] + if version > supported_schema_version: + raise RuntimeError("runtime schema is newer than this release; refusing downgrade") + if version < supported_schema_version: + pending = connection.execute("SELECT id FROM runs WHERE state NOT IN ('succeeded','failed','cancelled')").fetchall() + if pending: + raise RuntimeError( + "schema upgrade deferred: active/recoverable runs " + ", ".join(row[0] for row in pending) + + "; finish or cancel with the previous release, then retry" + ) + os.replace(temporary, selector) + if connection is not None: + connection.execute("COMMIT") + finally: + if connection is not None: + connection.close() diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 682e4cf..f916328 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -154,6 +154,7 @@ def __init__(self, database: Path, artifacts: Path): artifacts.mkdir(parents=True, exist_ok=True) self.connection = sqlite3.connect(database, timeout=10, isolation_level=None) self.connection.row_factory = sqlite3.Row + self.connection.create_function("devsquad_connection_schema", 0, lambda: SUPPORTED_SCHEMA_VERSION) try: self.connection.execute("PRAGMA busy_timeout=10000") self.connection.execute("PRAGMA foreign_keys=ON") @@ -172,6 +173,7 @@ def migrate(self) -> None: try: table = self.connection.execute("SELECT 1 FROM sqlite_master WHERE type='table' AND name='schema_migrations'").fetchone() current = self.connection.execute("SELECT COALESCE(MAX(version), 0) FROM schema_migrations").fetchone()[0] if table else 0 + initial_version = current if current > SUPPORTED_SCHEMA_VERSION: raise SchemaVersionError(f"database schema {current} is newer than supported {SUPPORTED_SCHEMA_VERSION}") if 0 < current < SUPPORTED_SCHEMA_VERSION: @@ -204,6 +206,24 @@ def migrate(self) -> None: ) self.connection.execute("INSERT INTO schema_migrations(version, applied_at) VALUES(?, ?)", (next_version, _utc_now())) current = next_version + if initial_version < SUPPORTED_SCHEMA_VERSION: + # An old Store opened before the exclusive upgrade can outlive + # the constructor's version check. Fence its later writes at + # the database boundary, including admission of a new run. + # Old packages either report their lower version or lack the + # function entirely; both fail before mutating a saved row. + tables = self.connection.execute( + "SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%' AND name!='schema_migrations'", + ).fetchall() + for table_row in tables: + name = table_row["name"].replace('"', '""') + for action in ("INSERT", "UPDATE", "DELETE"): + trigger = f"devsquad_schema_guard_{name}_{action.lower()}" + self.connection.execute( + f'CREATE TRIGGER IF NOT EXISTS "{trigger}" BEFORE {action} ON "{name}" BEGIN ' + "SELECT CASE WHEN devsquad_connection_schema() < (SELECT MAX(version) FROM schema_migrations) " + "THEN RAISE(ABORT, 'database schema is newer than connection supports') END; END" + ) self.connection.execute("COMMIT") except Exception: self.connection.execute("ROLLBACK") diff --git a/scripts/install-core.sh b/scripts/install-core.sh index d5bb63f..ed18173 100755 --- a/scripts/install-core.sh +++ b/scripts/install-core.sh @@ -391,27 +391,29 @@ PY RELEASE_CREATED=1 fi -# Migrate the explicitly scoped default ledger before switching the launcher. -# Store fences migrations transactionally against active/recoverable old runs. -# Custom runtimes receive the same guard on their first new-release operation. +# Never advance the ledger while the old release is still selected. Check +# upgrade readiness and replace the selector under one ledger lock; actual +# migration is lazy, through the new release's guarded Store constructor. UPGRADE_RUNTIME="${DEVSQUAD_RUNTIME_DIR:-${INSTALL_ROOT}/runtime}" case "$UPGRADE_RUNTIME" in /*) ;; *) fail "runtime directory must be absolute" ;; esac -if [ -f "$UPGRADE_RUNTIME/state.sqlite3" ]; then - "$RELEASE_DIR/venv/bin/python" -P - "$UPGRADE_RUNTIME" <<'PY' || fail "runtime upgrade deferred; previous current selector and launcher are unchanged" +activate_current() { + "$RELEASE_DIR/venv/bin/python" -P - "$1" "$INSTALL_ROOT/current" "$UPGRADE_RUNTIME" "$REPO_ROOT/plugin/core/src/devsquad/release_activation.py" <<'PY' || fail "runtime upgrade deferred; previous current selector and launcher are unchanged" from pathlib import Path +import runpy import sys -from devsquad.store import Store +from devsquad.store import SUPPORTED_SCHEMA_VERSION -runtime = Path(sys.argv[1]) +temporary, selector, runtime, helper = map(Path, sys.argv[1:]) +activate_release = runpy.run_path(str(helper))["activate_release"] try: - store = Store(runtime / "state.sqlite3", runtime / "artifacts") + activate_release(temporary, selector, runtime, supported_schema_version=SUPPORTED_SCHEMA_VERSION) except Exception as exc: + if temporary.is_symlink(): + temporary.unlink() print(str(exc), file=sys.stderr) raise SystemExit(1) -else: - store.close() PY -fi +} CURRENT_CHANGED=0 CURRENT_TARGET="releases/$RELEASE_ID" @@ -420,17 +422,11 @@ if [ -L "$INSTALL_ROOT/current" ] && [ "$(readlink "$INSTALL_ROOT/current")" = " elif [ -e "$INSTALL_ROOT/current" ] || [ -L "$INSTALL_ROOT/current" ]; then [ -L "$INSTALL_ROOT/current" ] || fail "current selector is not a symlink: $INSTALL_ROOT/current" ln -s "$CURRENT_TARGET" "$INSTALL_ROOT/.current.$$" - "$PYTHON" - "$INSTALL_ROOT/.current.$$" "$INSTALL_ROOT/current" <<'PY' -import os, sys -os.replace(sys.argv[1], sys.argv[2]) -PY + activate_current "$INSTALL_ROOT/.current.$$" CURRENT_CHANGED=1 else ln -s "$CURRENT_TARGET" "$INSTALL_ROOT/.current.$$" - "$PYTHON" - "$INSTALL_ROOT/.current.$$" "$INSTALL_ROOT/current" <<'PY' -import os, sys -os.replace(sys.argv[1], sys.argv[2]) -PY + activate_current "$INSTALL_ROOT/.current.$$" CURRENT_CHANGED=1 fi diff --git a/scripts/run-core-tests.py b/scripts/run-core-tests.py new file mode 100644 index 0000000..0114167 --- /dev/null +++ b/scripts/run-core-tests.py @@ -0,0 +1,39 @@ +#!/usr/bin/env python3 +"""Spawn-safe core gate with monotonic timing and SQLite finalizer diagnostics.""" + +from datetime import datetime, timezone +import gc +import json +from pathlib import Path +import sys +import time +import unittest + + +def main(): + root = Path(__file__).resolve().parents[1] + sys.path[:0] = [str(root / "plugin/core/src"), str(root / "test/core")] + unraisable = [] + original_hook = sys.unraisablehook + def record_unraisable(event): + unraisable.append(type(event.exc_value).__name__) + original_hook(event) + sys.unraisablehook = record_unraisable + started, utc = time.monotonic(), datetime.now(timezone.utc) + suite = unittest.defaultTestLoader.discover(str(root / "test/core")) + result = unittest.TextTestRunner( + verbosity=2 if "--verbose" in sys.argv else 1, + failfast="--failfast" in sys.argv, + ).run(suite) + gc.collect() + print(json.dumps({ + "tests": result.testsRun, "errors": len(result.errors), "failures": len(result.failures), + "skips": len(result.skipped), "unraisable": unraisable, + "monotonic_seconds": time.monotonic() - started, + "utc_seconds": (datetime.now(timezone.utc) - utc).total_seconds(), + })) + return 0 if result.wasSuccessful() and not unraisable else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index 10c7f10..398661c 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -8,13 +8,14 @@ import tempfile import threading import unittest +from unittest.mock import patch ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "plugin/core/src")) from devsquad.capacity import derive_pool_capacity, validate_observation from devsquad.contracts import ContractError -from devsquad.store import ConflictError, Store +from devsquad.store import ConflictError, SchemaVersionError, Store NOW = datetime(2026, 9, 27, 15, 0, tzinfo=timezone.utc) @@ -313,7 +314,17 @@ def test_schema_eight_active_attempt_is_backfilled_and_reconciled(self): connection.commit() connection.close() - store = Store(database, path / "artifacts") + with self.assertRaisesRegex(SchemaVersionError, "active/recoverable"): + Store(database, path / "artifacts") + # Keep the migration-9 SQL backfill unit coverage independently + # of public upgrades, which must now defer for old active runs. + connection = sqlite3.connect(database) + connection.executescript(next(migrations.glob("009_*.sql")).read_text()) + connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(9,?)", (NOW.isoformat(),)) + connection.commit() + connection.close() + with patch("devsquad.store.SUPPORTED_SCHEMA_VERSION", 9): + store = Store(database, path / "artifacts") self.addCleanup(store.close) self.assertEqual(store.active_pool_counts(), {"shared-pool": 1}) row = store.connection.execute( @@ -325,6 +336,10 @@ def test_schema_eight_active_attempt_is_backfilled_and_reconciled(self): (NOW.isoformat(),), ) self.assertEqual(store.active_pool_counts(), {}) + store.connection.execute("UPDATE runs SET state='failed' WHERE id='r'") + upgraded = Store(database, path / "artifacts") + self.addCleanup(upgraded.close) + self.assertEqual(upgraded.active_pool_counts(), {}) if __name__ == "__main__": diff --git a/test/core/test_install_core.py b/test/core/test_install_core.py index 3cf66a8..fdb657f 100644 --- a/test/core/test_install_core.py +++ b/test/core/test_install_core.py @@ -13,6 +13,7 @@ import time import unittest import zipfile +from unittest.mock import patch from devsquad_test_fixtures import branch_review_routing_documents @@ -515,8 +516,10 @@ def test_schema_13_update_defers_for_active_and_recoverable_old_runs(self): self.assertNotEqual(updated["current_target"], str(old_release)) self.assertTrue(old_release.is_dir()) with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: - self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) self.assertTrue(self.cli_json(launcher, "result", queued["run_id"], "--runtime-dir", str(runtime))["data"]["ready"]) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) finally: # Use the old package explicitly even if a later assertion fails. cleanup_script = "from pathlib import Path; import sys; from devsquad.service import Service; s=Service(Path(sys.argv[1])); [s.cancel(r) for r in sys.argv[2:]]" @@ -526,6 +529,82 @@ def test_schema_13_update_defers_for_active_and_recoverable_old_runs(self): subprocess.run([str(old_python), "-P", "-c", cleanup_script, str(runtime), live["run_id"], queued["run_id"]], capture_output=True, env=self.environment, cwd=self.root) + def test_failed_selector_activation_never_advances_the_old_ledger(self): + from devsquad.release_activation import activate_release + runtime = self.root / "activation-runtime" + runtime.mkdir() + database = runtime / "state.sqlite3" + with closing(sqlite3.connect(database)) as connection: + for migration in sorted((CORE / "src/devsquad/migrations").glob("*.sql")): + version = int(migration.name.split("_", 1)[0]) + if version > 13: + break + connection.executescript(migration.read_text()) + connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?, 'fixture')", (version,)) + connection.commit() + selector, temporary = self.root / "current", self.root / "prepared-selector" + selector.symlink_to("old-release") + temporary.symlink_to("new-release") + with patch("devsquad.release_activation.os.replace", side_effect=OSError("injected activation failure")): + with self.assertRaisesRegex(OSError, "injected activation failure"): + activate_release(temporary, selector, runtime, supported_schema_version=15) + self.assertEqual(os.readlink(selector), "old-release") + with closing(sqlite3.connect(database, timeout=1)) as connection: + connection.execute("BEGIN EXCLUSIVE") + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) + connection.rollback() + + def test_preopened_schema13_store_cannot_admit_work_after_new_schema_commit(self): + from devsquad.store import Store + old_source = self.root / "preopened-schema13-core" + old_source.mkdir() + archived = subprocess.run(["git", "archive", "f4fa657:plugin/core"], cwd=ROOT, + check=True, capture_output=True).stdout + with tarfile.open(fileobj=io.BytesIO(archived)) as archive: + archive.extractall(old_source, filter="data") + first = self.install(old_source) + old_python = Path(first["current_target"]) / "venv/bin/python" + runtime = self.install_root / "runtime" + repo = self.root / "preopened-repo" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + script = """ +import json,sys +from pathlib import Path +from devsquad.store import Store,git_common_dir +s=Store(Path(sys.argv[1])/'state.sqlite3',Path(sys.argv[1])/'artifacts') +s._project(git_common_dir(Path(sys.argv[2]))) +print('ready',flush=True) +sys.stdin.readline() +try: + s.claim_start(Path(sys.argv[2]),'late-old-client',{'task':'bounded'},'old-owner') +except Exception as exc: + print(json.dumps({'error':str(exc)}),flush=True) +else: + print(json.dumps({'unsafe_admission':True}),flush=True) +finally: + s.close() +""" + process = subprocess.Popen([str(old_python), "-P", "-c", script, str(runtime), str(repo)], + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + text=True, env=self.environment, cwd=self.root) + try: + self.assertEqual(process.stdout.readline().strip(), "ready") + self.install() + store = Store(runtime / "state.sqlite3", runtime / "artifacts") + try: + self.assertEqual(store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) + finally: + store.close() + output, stderr = process.communicate("admit\n", timeout=8) + self.assertEqual(process.returncode, 0, stderr) + self.assertIn("devsquad_connection_schema", json.loads(output)["error"]) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT COUNT(*) FROM runs").fetchone()[0], 0) + finally: + if process.poll() is None: + process.terminate() + process.communicate(timeout=5) + def cli_json(self, launcher, *arguments): result = subprocess.run( [str(launcher), *arguments, "--json"], From dc68944627919532539441ad6015e1ad68f9c942 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 15:50:45 -0700 Subject: [PATCH 151/197] WIP checkpoint: WIP checkpoint: Keep lifecycle fixture clocks current and preserve clean upgrade review (2026-10-01 15:50) --- docs/plans/engineering-team/RESUME.md | 24 +++++++++++ docs/plans/engineering-team/backlog.json | 4 +- .../R8-upgrade-review-2026-10-01.json | 16 +++++++- test/core/test_experiment_eligibility.py | 11 ++++- test/core/test_lifecycle.py | 40 +++++++++---------- 5 files changed, 70 insertions(+), 25 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index bd3f5d0..8cb5a65 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -4,6 +4,30 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ## Current position — October 1, 2026 +### Latest continuation — upgrade review passed, full-gate clock fixture repair + +At `6874de7`, the bounded native upgrade follow-up is clean. Both required +checks passed with verified unchanged candidate integrity; the accepted run +`3c89bb19-64ec-4600-9080-436df89edcfb` terminalized succeeded, version 22. +The proper spawn-safe full gate stopped failfast after 228 tests in 313.702s: +one lifecycle eligibility error, zero failures and zero unraisable diagnostics. +The original reader cause is `outcome observed_at exceeds allowed clock skew`. +The test injected module-import time, which ages beyond five minutes in the +full suite; a ten-minute-old clock reproduces it while the isolated case passes. +Use current runtime clocks in those fixtures; production validation remains +unchanged. An explicit regression keeps the stale-clock rejection strict. + +The focused lifecycle/eligibility gate passes 20 tests in 128.937 seconds; +all 227 Bash assertions, generated reference and whitespace checks pass. +Next: checkpoint, then one frozen full gate. Only after it passes update the actual installation and run Claude +delivery/handoff, Grok and updated Gemini/Antigravity proofs. No provider job is +currently live. The actual installation still points at the schema-13 release. + +Privacy warning: Antigravity's global native MCP listing unexpectedly printed +an unrelated StitchMCP credential. It is not repeated or saved in evidence; +the user was advised to rotate it. Capture/filter future listings to DevSquad +only. Do not alter unrelated credentials/settings or bypass the denied IDE UI. + ### Current continuation — R3 audit repairs and safe schema-update deferral The two independent findings now have source repairs and red regressions. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 1cc422b..263789e 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -21,8 +21,8 @@ "status": "partial", "artifact": "evidence/R3b2-eligibility-partial-2026-10-01.json", "scope": "Reader full gate passed at a4a87fd; schema-15 explicit revisions and shared lifecycle eligibility implemented with focused public fixtures; no install or provider calls", - "next_action": "Independent R3 audit found delivery evidence and unmeasured-ratio gaps; repairs plus bytecode isolation and schema-13 upgrade safety pass 37 affected tests and 227 Bash assertions. Checkpoint, run frozen full integration and bounded repair re-review, then safely refresh and prove requested live workflows.", - "limitations": "Full integration/re-review on audit repairs pending; R4 catalog/quota, R5 public controller/outcomes, R6 UX and R7 Council open. Installed schema 13 has no nonterminal runs but is not yet refreshed. Claude/Grok/updated Gemini live proofs pending." + "next_action": "Bounded upgrade re-review at 6874de7 is clean and accepted with two verified checks. Full gate stopped after 228 tests on an aged module-import clock in lifecycle fixtures; the strict-skew-preserving fixture repair passes 20 focused tests and 227 Bash assertions. Run one spawn-safe frozen full gate, then safely refresh and prove requested live workflows.", + "limitations": "Full integration after clock-fixture repair pending; R4 catalog/quota, R5 public controller/outcomes, R6 UX and R7 Council open. Installed schema 13 has no nonterminal runs but is not yet refreshed. Claude/Grok/updated Gemini live proofs pending." }, "planning_checkpoint": { "recorded_on": "2026-10-01", diff --git a/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json index 64298b5..ac0883c 100644 --- a/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json @@ -1,7 +1,7 @@ { "schema_version": 1, "reviewed_revision": "e8cdf07", - "status": "finding_repaired_targeted_full_and_re_review_pending", + "status": "upgrade_follow_up_pass_full_gate_fixture_repair_pending", "run_id": "fcc030af-26ec-423c-b462-b6fc0f8b55d1", "runtime": "/Users/Dikshant/.devsquad/private-probes/r3-bounded-repair-review-20261001", "review_sha256": "13b77ee0812c9beefda98df697af4189c132c7e311df995c7b78e9f6282b8644", @@ -17,6 +17,18 @@ "targeted": {"tests": 12, "seconds": 9.109, "result": "pass"}, "evidence_process_regressions": {"tests": 32, "seconds": 137.768, "result": "pass"}, "bash": {"assertions": 227, "files": 11, "result": "pass"}, - "full_gate": "Original stdin launch interrupted as invalid: multiprocessing cannot reopen ; repeat using tracked spawn-safe runner.", + "clock_fixture_gate": {"tests": 20, "seconds": 128.937, "result": "pass", "production_clock_skew_validation": "unchanged"}, + "full_gate": {"tests": 228, "seconds": 313.702, "errors": 1, "failures": 0, "unraisable": [], "result": "failed_failfast", "cause": "Lifecycle test injected module-import time more than five minutes before real outcome observations; reproduced with ten-minute-old clock. Production skew fence unchanged."}, + "upgrade_follow_up": { + "revision": "6874de7", "run_id": "3c89bb19-64ec-4600-9080-436df89edcfb", + "runtime": "/Users/Dikshant/.devsquad/private-probes/r8-upgrade-follow-up-20261001", + "reviewer": {"harness": "codex", "version": "0.159.2", "model": "gpt-6.1-sol", "effort": "low", "permission": "read_only", "verification": "verified"}, + "verdict": "clean", "required_checks": 2, "checks": "passed", "integrity": "verified_unchanged", + "host_disposition": "accept", "terminal": {"state": "succeeded", "version": 22}, + "review_sha256": "68d90b7b05a382b665df7862af528c97457451dc388f5140d39033d5c272fa2f", + "checks_sha256": "77de92c3ae4eec05188f956a995586d273ab10abd49eeaf693f4cee76fe495a3", + "review_attempt_sha256": "059d4286fbc991ec9cd169f7869a7ed68c2d3c2405ae247eaf18ac832d005f37", + "usage": {"input_tokens": 81225, "output_tokens": 675, "total_tokens": 81900, "worker_invocations": 1, "native_model_requests": null} + }, "installed_refresh": false } diff --git a/test/core/test_experiment_eligibility.py b/test/core/test_experiment_eligibility.py index 6a7b5cf..b3e2ea5 100644 --- a/test/core/test_experiment_eligibility.py +++ b/test/core/test_experiment_eligibility.py @@ -1,7 +1,7 @@ """New lifecycle authority is bound to current public saved-run evidence.""" import copy -from datetime import datetime, timezone +from datetime import datetime, timedelta, timezone import hashlib import json from pathlib import Path @@ -68,6 +68,15 @@ def test_unmeasured_ratios_cannot_bypass_finite_qualification_gates(self): with self.assertRaisesRegex(ContractError, "latency_ratio_missing.*usage_ratio_missing"): self.store.record_profile_qualification(qualification) + def test_live_outcomes_require_current_clock_not_module_import_time(self): + # A full suite may spend more than the permitted skew before this + # module's first lifecycle case. Keep the real clock fence strict. + with self.assertRaisesRegex(ContractError, "invalid_saved_run_evidence"): + self.store.record_profile_qualification( + self.qualification, now=datetime.now(timezone.utc) - timedelta(minutes=10), + ) + self.assertEqual(self.store.record_profile_qualification(self.qualification)["gate_failures"], []) + def next_fixture(self, experiment_id, *, profiles=None, candidate_succeeds=False): fixture = ExperimentRuntimeFixture( self.fixture.root / experiment_id, service=self.fixture.service, repo=self.fixture.repo, diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index 1d7be20..b743c82 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -256,16 +256,16 @@ def test_template_and_guarded_authority_boundaries_are_strict(self): def test_qualification_cas_promotion_and_rollback_are_replay_safe(self): template = lifecycle_template() - registered = self.store.register_profile_template(template, now=NOW) - replayed = self.store.register_profile_template(template, now=NOW) + registered = self.store.register_profile_template(template) + replayed = self.store.register_profile_template(template) self.assertFalse(registered["replayed"]) self.assertTrue(replayed["replayed"]) baseline = self.store.bootstrap_profile_binding( - template, self.incumbent, version=7, now=NOW, + template, self.incumbent, version=7, ) self.assertFalse(baseline["replayed"]) self.assertTrue(self.store.bootstrap_profile_binding( - template, self.incumbent, version=7, now=NOW, + template, self.incumbent, version=7, )["replayed"]) registry = { "schema_version": 1, @@ -288,11 +288,11 @@ def test_qualification_cas_promotion_and_rollback_are_replay_safe(self): ) evaluation = self.seed_experiment() qualified = self.store.record_profile_qualification( - self.qualification(evaluation), now=NOW, + self.qualification(evaluation), ) self.assertEqual(qualified["gate_failures"], []) self.assertTrue(self.store.record_profile_qualification( - self.qualification(evaluation), now=NOW, + self.qualification(evaluation), )["replayed"]) barrier = threading.Barrier(2) @@ -303,7 +303,7 @@ def promote(decision_id): try: barrier.wait(timeout=10) results.append(connection.change_profile_binding( - self.promotion(decision_id), now=NOW, + self.promotion(decision_id), )) except Exception as exc: results.append(exc) @@ -349,7 +349,7 @@ def promote(decision_id): ) winning_request = self.promotion(receipt["decision_id"]) self.assertTrue(self.store.change_profile_binding( - winning_request, now=NOW, + winning_request, )["replayed"]) regression = self.seed_experiment( @@ -371,7 +371,7 @@ def promote(decision_id): "reason": "Held-out regression requires the qualified predecessor.", "evidence_refs": ["experiment-profile-b-regression", "regression.json"], } - rolled_back = self.store.change_profile_binding(rollback, now=NOW) + rolled_back = self.store.change_profile_binding(rollback) self.assertEqual(rolled_back["receipt"]["to"]["binding_version"], 9) self.assertEqual(self.store.profile_binding("review.deep")["profile_id"], "profile-a") self.assertEqual(len(self.store.profile_binding_decisions("review.deep")), 2) @@ -400,7 +400,7 @@ def promote(decision_id): def test_insufficient_evidence_and_disabled_guarded_auto_cannot_promote(self): reviewed = lifecycle_template(update_mode="reviewed") self.store.bootstrap_profile_binding( - reviewed, self.incumbent, version=7, now=NOW, + reviewed, self.incumbent, version=7, ) incomplete = self.qualification({ "experiment": {"experiment_id": "unused"}, @@ -412,33 +412,33 @@ def test_insufficient_evidence_and_disabled_guarded_auto_cannot_promote(self): "verdict": "incomplete", "evidence_refs": [], }) - saved = self.store.record_profile_qualification(incomplete, now=NOW) + saved = self.store.record_profile_qualification(incomplete) self.assertIn("experiment_evidence_missing", saved["gate_failures"]) with self.assertRaisesRegex(ContractError, "qualified candidate"): self.store.change_profile_binding( - self.promotion("decision-insufficient"), now=NOW, + self.promotion("decision-insufficient"), ) evaluation = self.seed_experiment() self.store.record_profile_qualification( - self.qualification(evaluation), now=NOW, + self.qualification(evaluation), ) with self.assertRaisesRegex(ContractError, "not enabled"): self.store.change_profile_binding( - self.promotion("decision-auto", actor="guarded_auto"), now=NOW, + self.promotion("decision-auto", actor="guarded_auto"), ) def test_catalog_unavailable_incumbent_uses_only_qualified_predecessor(self): template = lifecycle_template() self.store.bootstrap_profile_binding( - template, self.incumbent, version=7, now=NOW, + template, self.incumbent, version=7, ) evaluation = self.seed_experiment() self.store.record_profile_qualification( - self.qualification(evaluation), now=NOW, + self.qualification(evaluation), ) self.store.change_profile_binding( - self.promotion("decision-promote-catalog"), now=NOW, + self.promotion("decision-promote-catalog"), ) registry = { "schema_version": 1, @@ -516,7 +516,7 @@ def test_catalog_unavailable_incumbent_uses_only_qualified_predecessor(self): ) self.store.change_profile_binding( - self.promotion("decision-repromote-catalog", expected=9), now=NOW, + self.promotion("decision-repromote-catalog", expected=9), ) all_removed_path = self.root / "catalog-all-removed.json" update_last_good( @@ -535,7 +535,7 @@ def test_catalog_unavailable_incumbent_uses_only_qualified_predecessor(self): "catalog_change": all_removed, } with self.assertRaisesRegex(ContractError, "no available qualified predecessor"): - self.store.fallback_unavailable_profile_binding(blocked, now=NOW) + self.store.fallback_unavailable_profile_binding(blocked) self.assertEqual( self.store.profile_binding("review.deep")["profile_id"], "profile-b", ) @@ -545,7 +545,7 @@ def test_bootstrap_requires_a_proven_baseline(self): trial["quality_status"] = "trial" with self.assertRaisesRegex(ContractError, "already be proven"): self.store.bootstrap_profile_binding( - lifecycle_template(), trial, version=1, now=NOW, + lifecycle_template(), trial, version=1, ) def test_schema_twelve_contains_lifecycle_ledger(self): From c7ccc0245abe475f25fdd018d42a8fa2700cb0c0 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 15:59:58 -0700 Subject: [PATCH 152/197] WIP checkpoint: Record 455-test repair gate before safe installed workflow proofs (2026-10-01 15:59) --- docs/plans/engineering-team/RESUME.md | 14 +++++++++++--- docs/plans/engineering-team/backlog.json | 4 ++-- .../evidence/R8-upgrade-review-2026-10-01.json | 3 ++- 3 files changed, 15 insertions(+), 6 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 8cb5a65..9b410f3 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -19,9 +19,17 @@ unchanged. An explicit regression keeps the stale-clock rejection strict. The focused lifecycle/eligibility gate passes 20 tests in 128.937 seconds; all 227 Bash assertions, generated reference and whitespace checks pass. -Next: checkpoint, then one frozen full gate. Only after it passes update the actual installation and run Claude -delivery/handoff, Grok and updated Gemini/Antigravity proofs. No provider job is -currently live. The actual installation still points at the schema-13 release. +The frozen full gate at `dc68944` now passes **455 tests in 485.168 seconds**, +with two optional-SDK skips, zero errors/failures and zero unraisable diagnostics. +UTC 485.364056s and monotonic 485.360219s agree. Retain the failed 228-test +clock-fixture gate as history; no additional production validation was relaxed. +Next: checkpoint, safely refresh the actual installation using the existing +Python 3.12/offline MCP wheelhouse, run installed SDK tests and prove Claude +delivery/handoff, Grok and updated Gemini/Antigravity. Use a bounded genuine G4 +check-discovery repair for the Claude implementation → Codex review → tests +workflow. The private offline probe reproduces both wrong-checkout discovery +and missing Python core checks. No provider job is currently live. The actual +installation still points at the schema-13 release. Privacy warning: Antigravity's global native MCP listing unexpectedly printed an unrelated StitchMCP credential. It is not repeated or saved in evidence; diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 263789e..8a0b15f 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -21,8 +21,8 @@ "status": "partial", "artifact": "evidence/R3b2-eligibility-partial-2026-10-01.json", "scope": "Reader full gate passed at a4a87fd; schema-15 explicit revisions and shared lifecycle eligibility implemented with focused public fixtures; no install or provider calls", - "next_action": "Bounded upgrade re-review at 6874de7 is clean and accepted with two verified checks. Full gate stopped after 228 tests on an aged module-import clock in lifecycle fixtures; the strict-skew-preserving fixture repair passes 20 focused tests and 227 Bash assertions. Run one spawn-safe frozen full gate, then safely refresh and prove requested live workflows.", - "limitations": "Full integration after clock-fixture repair pending; R4 catalog/quota, R5 public controller/outcomes, R6 UX and R7 Council open. Installed schema 13 has no nonterminal runs but is not yet refreshed. Claude/Grok/updated Gemini live proofs pending." + "next_action": "Full spawn-safe integration at dc68944 passes 455 tests (two optional-SDK skips, zero errors/failures/unraisable diagnostics) after the lifecycle clock fixture repair. Bounded upgrade review is clean. Safely refresh using the existing Python 3.12/MCP wheelhouse, verify installed SDK, then prove Claude delivery/handoff, Grok and updated Gemini/Antigravity.", + "limitations": "R4 catalog/quota, R5 public controller/outcomes, R6 UX and R7 Council open. Installed schema 13 has no nonterminal runs but is not yet refreshed. Claude/Grok/updated Gemini live proofs pending." }, "planning_checkpoint": { "recorded_on": "2026-10-01", diff --git a/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json index ac0883c..01a72ba 100644 --- a/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-upgrade-review-2026-10-01.json @@ -1,7 +1,7 @@ { "schema_version": 1, "reviewed_revision": "e8cdf07", - "status": "upgrade_follow_up_pass_full_gate_fixture_repair_pending", + "status": "upgrade_follow_up_and_full_gate_pass_installed_proofs_pending", "run_id": "fcc030af-26ec-423c-b462-b6fc0f8b55d1", "runtime": "/Users/Dikshant/.devsquad/private-probes/r3-bounded-repair-review-20261001", "review_sha256": "13b77ee0812c9beefda98df697af4189c132c7e311df995c7b78e9f6282b8644", @@ -18,6 +18,7 @@ "evidence_process_regressions": {"tests": 32, "seconds": 137.768, "result": "pass"}, "bash": {"assertions": 227, "files": 11, "result": "pass"}, "clock_fixture_gate": {"tests": 20, "seconds": 128.937, "result": "pass", "production_clock_skew_validation": "unchanged"}, + "full_gate_after_fixture_repair": {"revision": "dc68944", "tests": 455, "seconds": 485.168, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "monotonic_seconds": 485.3602193329716, "utc_seconds": 485.364056, "result": "pass", "command": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 scripts/run-core-tests.py --failfast"}, "full_gate": {"tests": 228, "seconds": 313.702, "errors": 1, "failures": 0, "unraisable": [], "result": "failed_failfast", "cause": "Lifecycle test injected module-import time more than five minutes before real outcome observations; reproduced with ten-minute-old clock. Production skew fence unchanged."}, "upgrade_follow_up": { "revision": "6874de7", "run_id": "3c89bb19-64ec-4600-9080-436df89edcfb", From ea113d49f913dd760950c5ad230a076b511d8e11 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 16:02:02 -0700 Subject: [PATCH 153/197] WIP checkpoint: Record safe installed schema-15 update and SDK regression proof (2026-10-01 16:02) --- docs/plans/engineering-team/RESUME.md | 14 +++++++++++ .../R8-installed-workflows-2026-10-01.json | 25 +++++++++++++++++++ 2 files changed, 39 insertions(+) create mode 100644 docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 9b410f3..0931a34 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,20 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Latest continuation — upgrade review passed, full-gate clock fixture repair +The actual local installation is now safely refreshed to +`0.1.0-py31214-674889018f98-mcp-a26bc88afbef` using the existing Python 3.12.14 +and locked offline MCP 2.2.0 wheelhouse. Reinstall is unchanged, all payload +drift flags are false, `pip check` passes, all four registrations match without +changes, and **22 installed-SDK tests pass in 5.585s with no skips**. The +schema-13 ledger had no active runs; a private SQLite backup and the old release +are retained. First new-release access migrated to schema 15 and read an old +succeeded result unchanged. See [installed workflow evidence](evidence/R8-installed-workflows-2026-10-01.json). +Next: genuine bounded G4 repair through installed `squad fix`, Claude writer, +independent Codex review and mandatory core checks; then actual Claude MCP +handoff plus Grok/Gemini status probes against that same updated installation. +Do not repeat the unchanged full gate or upgrade review. No model probe is +currently live; record its run ID before a handoff/interruption. + At `6874de7`, the bounded native upgrade follow-up is clean. Both required checks passed with verified unchanged candidate integrity; the accepted run `3c89bb19-64ec-4600-9080-436df89edcfb` terminalized succeeded, version 22. diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json new file mode 100644 index 0000000..a7f9d91 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -0,0 +1,25 @@ +{ + "schema_version": 1, + "status": "installed_regressions_pass_live_workflows_pending", + "source_checkpoint": "c7ccc02", + "installation": { + "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", + "release": "0.1.0-py31214-674889018f98-mcp-a26bc88afbef", + "source_digest": "674889018f980ce648c6ac349d46253a723698206f8c6994454aa947e6efd4ab", + "source_plugin_installed_drift": false, + "python": "3.12.14", "mcp": "2.2.0", "network_downloads": false, + "first_install_changed": true, "reinstall_changed": false, + "pip_check": "pass", "previous_release_retained": true, + "ledger_backup": "/Users/Dikshant/.devsquad/backups/runtime-before-schema15-20261001.sqlite3", + "schema_before": 13, "schema_after_first_access": 15, + "nonterminal_runs_before": 0, + "old_result_readback": {"run_id": "c611ad4c-5473-4f05-a870-a136f24464d3", "state": "succeeded", "version": 22}, + "setup": {"hosts": ["antigravity", "claude-code", "codex", "grok"], "matching": 4, "changed": 0}, + "sdk_tests": {"tests": 22, "seconds": 5.585, "skips": 0, "result": "pass"} + }, + "delivery": {"status": "pending", "scope": "G4 exact-target check discovery and inclusion of Python core regressions"}, + "claude_handoff": {"status": "pending"}, + "grok_operation": {"status": "pending"}, + "gemini_antigravity": {"status": "pending", "ide_ui": "computer_use_permission_denied_do_not_bypass"}, + "limitations": ["R4 catalog/quota, R5 public trials/outcomes, remaining R6 UX and R7 Council are separate open scope", "Jev runtime remains off and its one-request allowance is spent", "No paid API fallback, purchases, resets, global AI settings, push or external messages"] +} From 3555a91b8730044d9f3f60930589b3ddfc9e1165 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 16:15:08 -0700 Subject: [PATCH 154/197] WIP checkpoint: Repair Claude argument framing and preserve correlated stream identity with auxiliary usage (2026-10-01 16:15) --- docs/plans/engineering-team/CONTRACTS.md | 13 +++- docs/plans/engineering-team/RESUME.md | 28 +++++++ .../R8-installed-workflows-2026-10-01.json | 9 ++- plugin/core/src/devsquad/adapters.py | 2 +- .../src/devsquad/claude_delivery_worker.py | 9 ++- plugin/core/src/devsquad/claude_identity.py | 77 ++++++++++++++++--- test/core/test_adapters.py | 1 + test/core/test_claude_identity.py | 66 ++++++++++++++++ test/core/test_delivery_workflow.py | 6 ++ 9 files changed, 186 insertions(+), 25 deletions(-) diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index c67cedc..ba627be 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -83,11 +83,16 @@ Install-time discovery reports supported values and evidence (`documented`, `pro Claude implementation evidence v2 keeps the frozen requested profile separate from `observed_identity`. The tested native result must have a success envelope, -bounded session ID and a single concrete `modelUsage` entry. Its model key is -the reported serving identity; `canonicalModel` is pricing metadata only. +bounded session ID and a concrete reported writer identity. Legacy native JSON +requires a single `modelUsage` entry; its key is the reported identity. Native +stream evidence v2 instead requires all top-level assistant messages to report +one concrete model under the final result's session, with that model also +present in terminal usage. Auxiliary usage entries remain visible and cannot +substitute for the writer; `canonicalModel` is pricing metadata only. Tested family aliases may resolve to a reported concrete member of that family, -without hardcoded current revisions. Multiple entries cannot identify a unique -writer and fail closed. Effective effort and backing revision remain null when +without hardcoded current revisions. Multiple usage-only entries, missing or +contradictory stream model/session reports, nested delegated messages and a +writer absent from terminal usage fail closed. Effective effort and backing revision remain null when the result does not report them; `verification_scope: reported_model` does not claim that these unknown settings were verified. Native session, typed usage, alias resolution and the normalized result evidence are retained and validated diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 0931a34..735c327 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -4,6 +4,34 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ## Current position — October 1, 2026 +### Latest continuation — Claude launch/stream repair, Grok and Gemini proof + +The installed G4 delivery attempt at `ea113d4`, run +`8a8a4548-13c1-4a2f-9190-870112f63ef2`, failed before implementation, version +13, with zero native reported tokens. The real cause is the variadic Claude +`--tools` option consuming a trailing prompt. Both launch paths now add `--`; +two red regressions reproduced the problem before repair. The tool-free native +smoke succeeds with that framing, but current Claude reports auxiliary Haiku +usage alongside the Sonnet writer. Strict stream evidence now requires all +top-level assistant messages to identify one concrete model under the final +session, retains all auxiliary usage, and revalidates on import. Usage-only +multi-model JSON still fails closed. Contracts updated; **62 focused tests pass +in 70.381s**, 227 Bash assertions and generated reference/whitespace pass. +Next: checkpoint, bounded independent Claude framing/identity review plus one +frozen full gate; refresh to these repairs before retrying G4 under a new run. +The actual installed release is still `674889018f98` and predates this repair. + +Grok 0.2.111 was refused with HTTP 426. The user explicitly authorized its +normal local update; 1.0.46 stable is installed with the old executable backed +up privately. Native `grok-4.7-build` actually called DevSquad status through +MCP and observed the exact failed run/version 13. Two bounded successful status +smokes are recorded, not hidden; no API fallback/settings changes. Gemini +3.8 Flash Low via Antigravity 1.2.13 also actually called the same updated +installation's MCP status and observed that exact run/version. Its first +unitless timeout was rejected at argument parsing; the corrected 120s call +succeeded. IDE UI remains permission-denied, distinct from CLI/MCP proof. +These proofs are in [installed workflow evidence](evidence/R8-installed-workflows-2026-10-01.json). + ### Latest continuation — upgrade review passed, full-gate clock fixture repair The actual local installation is now safely refreshed to diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index a7f9d91..b2a54b8 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "status": "installed_regressions_pass_live_workflows_pending", + "status": "grok_gemini_pass_claude_launch_repairs_pending_full_gate", "source_checkpoint": "c7ccc02", "installation": { "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", @@ -17,9 +17,10 @@ "setup": {"hosts": ["antigravity", "claude-code", "codex", "grok"], "matching": 4, "changed": 0}, "sdk_tests": {"tests": 22, "seconds": 5.585, "skips": 0, "result": "pass"} }, - "delivery": {"status": "pending", "scope": "G4 exact-target check discovery and inclusion of Python core regressions"}, + "delivery": {"status": "initial_attempt_failed_before_implementation", "run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13, "worker_invocations": 1, "reported_tokens": 0, "cause": "Claude variadic --tools consumes the final positional prompt without an explicit -- separator", "scope": "G4 exact-target check discovery and inclusion of Python core regressions"}, + "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, "claude_handoff": {"status": "pending"}, - "grok_operation": {"status": "pending"}, - "gemini_antigravity": {"status": "pending", "ide_ui": "computer_use_permission_denied_do_not_bypass"}, + "grok_operation": {"status": "pass", "old_version_blocker": "HTTP 426: 0.2.111 rejected, requires 1.0.13 or later", "user_authorized_update": true, "updated_version": "1.0.46 (2765805b9442) [stable]", "old_executable_backup": "private_probe_directory", "native_model": "grok-4.7-build", "tool": "devsquad__squad_status", "observed": {"run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13}, "stdout_sha256": "bc031b922e8c36b3cdfff5c7c717916a27019a19d650f67cd7402b832544bfdd", "stderr_sha256": "1be68163f727c0712fefcd45b80d5f1d383ee995ff8f61f7da601fe82ee4a9b3", "usage": {"input_tokens": 34088, "cache_read_input_tokens": 69888, "output_tokens": 434, "reasoning_tokens": 285, "total_tokens": 104410, "turns": 3}, "extra_bounded_smoke": {"result": "pass", "stdout_sha256": "cbc772baa3dabf93b833cf9843c5a873bb72cbebf9fb8ddf7d9b5ae7f12a8f43", "total_tokens": 103523}, "scope": "native Grok Build MCP operation, not a verified implementation/reviewer adapter"}, + "gemini_antigravity": {"status": "pass_cli_mcp", "version": "1.2.13", "native_model": "gemini-3.8-flash-low", "tool": "devsquad/squad_status", "permission": "existing project-scoped status grant, plan mode and sandbox", "schema_discovery": "read only DevSquad squad_status tool schema", "observed": {"run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13}, "seconds": 10.62698745902162, "stdout_sha256": "8781e538501b3553c100ca754d2880c8f45a011062eddd5fd3b108cf0878181d", "usage": {"input_tokens": 33060, "output_tokens": 220, "thinking_tokens": 0, "cache_read_tokens": 44852, "total_tokens": 33280}, "initial_probe": "Argument parser rejected unitless print timeout before generation; corrected to 120s", "ide_ui": "computer_use_permission_denied_do_not_bypass"}, "limitations": ["R4 catalog/quota, R5 public trials/outcomes, remaining R6 UX and R7 Council are separate open scope", "Jev runtime remains off and its one-request allowance is spent", "No paid API fallback, purchases, resets, global AI settings, push or external messages"] } diff --git a/plugin/core/src/devsquad/adapters.py b/plugin/core/src/devsquad/adapters.py index ace6d04..87f759a 100644 --- a/plugin/core/src/devsquad/adapters.py +++ b/plugin/core/src/devsquad/adapters.py @@ -163,7 +163,7 @@ def prepare_cli( if effort: args += ["--effort", effort] args.extend(permission_args) - args += [prompt] + args += ["--", prompt] permission_args = () else: raise ContractError(f"no argv builder for adapter: {manifest.name}") diff --git a/plugin/core/src/devsquad/claude_delivery_worker.py b/plugin/core/src/devsquad/claude_delivery_worker.py index 437b7aa..9318ec1 100644 --- a/plugin/core/src/devsquad/claude_delivery_worker.py +++ b/plugin/core/src/devsquad/claude_delivery_worker.py @@ -157,11 +157,11 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: snapshot.get("revision_request"), ) argv = [ - str(binary), "--print", "--output-format", "json", "--safe-mode", + str(binary), "--print", "--output-format", "stream-json", "--verbose", "--safe-mode", "--disable-slash-commands", "--no-session-persistence", "--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}', "--no-chrome", "--model", model, "--effort", effort, - *adapter["permission_args"], prompt, + *adapter["permission_args"], "--", prompt, ] timeout_seconds = snapshot["task"]["budget"]["wall_seconds"] try: @@ -203,8 +203,9 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: error_code = "CLI_ERROR" provider_document = None try: - provider_document = strict_json(stdout) - summary, native = native_result(provider_document) + from .claude_identity import decode_native_result + provider_document, writer_messages = decode_native_result(stdout) + summary, native = native_result(provider_document, writer_messages) observed = observed_identity(native, adapter, profile) except ContractError as exc: safe_document = provider_document if isinstance(provider_document, dict) else {} diff --git a/plugin/core/src/devsquad/claude_identity.py b/plugin/core/src/devsquad/claude_identity.py index 54a7b40..681a4dd 100644 --- a/plugin/core/src/devsquad/claude_identity.py +++ b/plugin/core/src/devsquad/claude_identity.py @@ -1,7 +1,8 @@ """Bounded Claude result evidence; requested settings are never observations. -The tested CLI result attributes usage to model IDs, not individual writer -messages. Only a single concrete reported model can establish writer identity. +Legacy JSON results need a single concrete usage model. Native streams can +identify a unique writer through correlated assistant-message model reports, +while retaining separately reported auxiliary usage without guessing its role. Pricing aliases are metadata. Effective effort and serving revision are unknown. """ @@ -138,7 +139,42 @@ def model_usage(raw: Any) -> dict[str, Any]: return result -def native_result(document: Any) -> tuple[str, dict[str, Any]]: +def decode_native_result(payload: bytes | str) -> tuple[dict[str, Any], dict[str, Any] | None]: + """Decode strict legacy JSON or a bounded, session-correlated native stream.""" + try: + document = strict_json(payload) + except ContractError: + encoded = payload.encode("utf-8") if isinstance(payload, str) else payload + if not isinstance(encoded, bytes) or len(encoded) > MAX_OUTPUT_BYTES: + raise ContractError("Claude stream exceeds its bound") + records = [strict_json(line) for line in encoded.splitlines() if line.strip()] + if not 1 <= len(records) <= 8192 or any(not isinstance(item, dict) for item in records): + raise ContractError("Claude stream records are invalid") + terminals = [item for item in records if item.get("type") == "result"] + if len(terminals) != 1 or records[-1] is not terminals[0]: + raise ContractError("Claude stream requires one final result") + document = terminals[0] + session = _session(document.get("session_id")) + models = [] + for item in records: + if item.get("type") != "assistant": + continue + message = item.get("message") + if (item.get("session_id") != session or item.get("parent_tool_use_id") is not None + or not isinstance(message, dict) or message.get("role") != "assistant"): + raise ContractError("Claude writer message is not session-correlated") + models.append(_model(message.get("model"))) + if not models: + if document.get("is_error") is True: + return document, None + raise ContractError("Claude stream has no reported writer messages") + return document, {"session_id": session, "models": sorted(set(models)), "message_count": len(models)} + if not isinstance(document, dict): + raise ContractError("Claude result must be an object") + return document, None + + +def native_result(document: Any, writer_messages: dict[str, Any] | None = None) -> tuple[str, dict[str, Any]]: if (not isinstance(document, dict) or document.get("type") != "result" or document.get("subtype") != "success" or document.get("is_error") is not False @@ -149,27 +185,45 @@ def native_result(document: Any) -> tuple[str, dict[str, Any]]: if len(summary) > 20_000: raise ContractError("Claude summary exceeds its limit") return summary, { - "schema_version": 1, "type": "result", "subtype": "success", + "schema_version": 1 if writer_messages is None else 2, "type": "result", "subtype": "success", "is_error": False, "session_id": _session(document.get("session_id")), "model_usage": model_usage(document.get("modelUsage")), "top_level_model": _model(document["model"]) if "model" in document else None, "usage": reported_usage(document.get("usage")), + **({"writer_messages": writer_messages} if writer_messages is not None else {}), } def observed_identity(native: Any, adapter: dict[str, Any], profile: dict[str, Any]) -> dict[str, Any]: - _exact(native, _NATIVE_FIELDS) - if (type(native["schema_version"]) is not int or native["schema_version"] != 1 + if not isinstance(native, dict): + raise ContractError("Claude native identity envelope is invalid") + version = native.get("schema_version") + _exact(native, _NATIVE_FIELDS | ({"writer_messages"} if version == 2 else set())) + if (type(version) is not int or version not in {1, 2} or native["type"] != "result" or native["subtype"] != "success" or native["is_error"] is not False): raise ContractError("Claude native identity envelope is invalid") _session(native["session_id"]) _usage_evidence(native["usage"]) models = model_usage(native["model_usage"]) - if models != native["model_usage"] or len(models) != 1: - raise ContractError("Claude result cannot identify a unique writer model") - model = next(iter(models)) + if models != native["model_usage"]: + raise ContractError("Claude native model usage is inconsistent") + if version == 1: + if len(models) != 1: + raise ContractError("Claude result cannot identify a unique writer model") + model = next(iter(models)) + model_source = "claude.result.modelUsage" + else: + messages = _exact(native["writer_messages"], {"session_id", "models", "message_count"}) + if (messages["session_id"] != native["session_id"] + or type(messages["message_count"]) is not int or not 1 <= messages["message_count"] <= 8192 + or not isinstance(messages["models"], list) or len(messages["models"]) != 1): + raise ContractError("Claude stream cannot identify a unique correlated writer model") + model = _model(messages["models"][0]) + if model not in models: + raise ContractError("Claude stream writer is missing from terminal model usage") + model_source = "claude.stream.assistant.message.model" if model in _ALIASES: raise ContractError("Claude reported identity is an unresolved alias") requested = _model(profile.get("model_id")) @@ -187,8 +241,7 @@ def observed_identity(native: Any, adapter: dict[str, Any], or profile.get("harness") != "claude" or profile.get("permission_policy") != "workspace_write"): raise ContractError("Claude identity does not match the frozen adapter") - provider = models[model].get("provider") - if provider is not None and provider != adapter["model_provider"]: + if any(entry.get("provider", adapter["model_provider"]) != adapter["model_provider"] for entry in models.values()): raise ContractError("Claude reported provider contradicts the frozen adapter") return { "harness": "claude", "harness_version": adapter["harness_version"], @@ -196,7 +249,7 @@ def observed_identity(native: Any, adapter: dict[str, Any], "effort": None, "backing_revision": None, "permission_policy": "workspace_write", "verification": "verified", "verification_scope": "reported_model", - "model_source": "claude.result.modelUsage", + "model_source": model_source, "alias_resolution": {"requested": requested, "reported": model} if alias else None, "native_evidence": native, } diff --git a/test/core/test_adapters.py b/test/core/test_adapters.py index 0a0cda6..0d7fc0c 100644 --- a/test/core/test_adapters.py +++ b/test/core/test_adapters.py @@ -60,6 +60,7 @@ def test_read_only_argv_is_structured_and_has_no_bypass_or_delegation(self): self.assertIn("Read,Glob,Grep", spec.argv) self.assertIn('{"mcpServers":{}}', spec.argv) self.assertIn("Fix src/My Parser.ts without delegating.", spec.argv) + self.assertEqual(spec.argv[-2:], ("--", "Fix src/My Parser.ts without delegating.")) self.assertNotIn("--dangerously-skip-permissions", spec.argv) self.assertNotIn("Agent", ",".join(spec.argv)) self.assertNotIn("Bash", ",".join(spec.argv)) diff --git a/test/core/test_claude_identity.py b/test/core/test_claude_identity.py index 37f19fc..cd87f8d 100644 --- a/test/core/test_claude_identity.py +++ b/test/core/test_claude_identity.py @@ -206,6 +206,72 @@ def test_multiple_model_entries_do_not_identify_a_unique_writer(self): with self.assertRaises(ContractError): self.run_document(document) + def stream_records(self): + document = self.document() + document["modelUsage"]["claude-haiku-4-5-20251001"] = self.model_usage() + return [{ + "type": "assistant", "session_id": document["session_id"], + "parent_tool_use_id": None, + "message": {"role": "assistant", "model": self.MODEL}, + }, document] + + def run_stream(self, records, *, requested_model=None): + self.output.write_text("\n".join(json.dumps(item) for item in records) + "\n") + return run_claude_implementer(self.snapshot(requested_model)) + + def test_correlated_stream_identifies_writer_and_retains_auxiliary_usage(self): + evidence = self.run_stream(self.stream_records(), requested_model="sonnet") + attempt = evidence["attempt"] + observed = attempt["observed_identity"] + self.assertEqual(observed["model_id"], self.MODEL) + self.assertEqual(observed["model_source"], "claude.stream.assistant.message.model") + native = observed["native_evidence"] + self.assertEqual(native["schema_version"], 2) + self.assertEqual(native["writer_messages"], { + "session_id": "session-identity-fixture", "models": [self.MODEL], "message_count": 1, + }) + self.assertEqual(len(native["model_usage"]), 2) + self.assertIsNone(observed["effort"]) + self.assertIsNone(observed["backing_revision"]) + + def test_stream_cannot_guess_writer_from_usage_or_requested_settings(self): + import copy + original = self.stream_records() + mutations = [] + for path, value in ( + ((0, "session_id"), "unrelated-session"), + ((0, "parent_tool_use_id"), "delegated-tool"), + ((0, "message", "model"), "claude-other"), + ((0, "message", "role"), "user"), + ((0, "message", "model"), None), + ((1, "modelUsage"), {"claude-haiku-4-5-20251001": self.model_usage()}), + ): + records = copy.deepcopy(original) + target = records + for key in path[:-1]: + target = target[key] + target[path[-1]] = value + mutations.append(records) + contradictory = copy.deepcopy(original[0]) + contradictory["message"]["model"] = "claude-haiku-4-5-20251001" + mutations += [original[1:], [original[0], contradictory, original[1]], + [*original, original[1]], [*original, {"type": "system"}]] + for index, records in enumerate(mutations): + with self.subTest(case=index), self.assertRaises(ContractError): + self.run_stream(records) + + def test_stream_session_model_proof_is_revalidated_on_import(self): + import copy + from devsquad.workflows import validate_implementation_evidence + snapshot = self.snapshot() + evidence = self.run_stream(self.stream_records()) + for field, value in (("session_id", "other"), ("models", [self.MODEL, "claude-haiku"]), + ("message_count", True), ("message_count", 0)): + changed = copy.deepcopy(evidence) + changed["attempt"]["observed_identity"]["native_evidence"]["writer_messages"][field] = value + with self.subTest(field=field, value=value), self.assertRaises(ContractError): + validate_implementation_evidence(changed, snapshot) + def test_family_alias_resolves_only_to_reported_concrete_model(self): evidence = self.run_document(self.document(), requested_model="sonnet") attempt = evidence["attempt"] diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py index 6233c44..d5e67d6 100644 --- a/test/core/test_delivery_workflow.py +++ b/test/core/test_delivery_workflow.py @@ -383,6 +383,12 @@ def test_frozen_claude_worker_edits_only_the_delivery_workspace(self): " printf '%s\\n' '2.1.220 (Claude Code)'\n" " exit 0\n" "fi\n" + "separator=0\n" + "for argument do\n" + " if [ \"$separator\" = 1 ]; then break; fi\n" + " if [ \"$argument\" = -- ]; then separator=1; fi\n" + "done\n" + "[ \"$separator\" = 1 ] || exit 9\n" "printf '%s\\n' \"VALUE = 'fixed'\" > src/app.py\n" "printf '%s\\n' '{\"type\":\"result\",\"subtype\":\"success\",\"is_error\":false," "\"result\":\"Applied the bounded fix.\"," From 6f51db981333fbf50d23f94ebf7800889a39a99d Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 16:28:05 -0700 Subject: [PATCH 155/197] WIP checkpoint: Map verified Claude first-party transport labels and record real host handoff (2026-10-01 16:28) --- docs/plans/engineering-team/CONTRACTS.md | 3 +++ docs/plans/engineering-team/RESUME.md | 23 +++++++++++++++++++ .../R8-installed-workflows-2026-10-01.json | 7 ++++-- plugin/core/src/devsquad/claude_identity.py | 7 +++++- test/core/test_claude_identity.py | 17 ++++++++++++++ 5 files changed, 54 insertions(+), 3 deletions(-) diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index ba627be..af5ca58 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -97,6 +97,9 @@ the result does not report them; `verification_scope: reported_model` does not claim that these unknown settings were verified. Native session, typed usage, alias resolution and the normalized result evidence are retained and validated again on coordinator import, before candidate finalization. +The tested Claude 2.1.220 transport label `firstParty` maps to Anthropic only +for that verified harness version; the original label remains in evidence. +Unknown or contradictory provider labels still block identity verification. New delivery review imports and acceptance require verified different reported model IDs, regardless of requested aliases, family labels or harness names. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 735c327..ba58ef6 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,29 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Latest continuation — Claude launch/stream repair, Grok and Gemini proof +The frozen stream/framing gate at `3555a91` passes **458 tests in 467.018s**, +two optional-SDK skips, zero errors/failures/unraisable diagnostics. Independent +native Codex review is clean with all three declared checks passed. A real +Claude Code MCP lead claimed, renewed and completed that saved review run +`b9501b55-1c65-4b32-a03f-1087cca8fefc`: **succeeded, version 23**. Earlier +probe failures are retained: empty builtins left MCP pending; a later model +falsely alleged a hash mismatch. Exact machine comparison confirmed all refs +matched before the successful fenced completion. ToolSearch plus only the +three scoped MCP tools works, without file/shell/delegation access. + +The real native stream additionally labels provider `firstParty`. It now maps +to Anthropic only for verified Claude 2.1.220, preserving its raw label and +rejecting other/unverified provider labels. The **63-test focused gate passes +in 70.391s**, 227 Bash assertions/reference/diff pass, and the saved real stream +now decodes to verified Sonnet 5 with Haiku usage retained. This mapping follows +the 458-test gate; the mandatory full suite on the upcoming genuine G4 candidate +must include it, and that independent delivery review will inspect the mapping. +Next: checkpoint, refresh the actual installation to these tested launch +repairs, retry G4 in a new saved run, and finish candidate review/checks plus +host disposition. Do not repeat Grok/Gemini or the completed Claude handoff. +Those live MCP proofs used the first refreshed release; MCP service source is +unchanged in the subsequent Claude worker repairs. Broader R4–R7 remain open. + The installed G4 delivery attempt at `ea113d4`, run `8a8a4548-13c1-4a2f-9190-870112f63ef2`, failed before implementation, version 13, with zero native reported tokens. The real cause is the variadic Claude diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index b2a54b8..42ab9be 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "status": "grok_gemini_pass_claude_launch_repairs_pending_full_gate", + "status": "claude_handoff_grok_gemini_pass_managed_delivery_pending", "source_checkpoint": "c7ccc02", "installation": { "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", @@ -19,7 +19,10 @@ }, "delivery": {"status": "initial_attempt_failed_before_implementation", "run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13, "worker_invocations": 1, "reported_tokens": 0, "cause": "Claude variadic --tools consumes the final positional prompt without an explicit -- separator", "scope": "G4 exact-target check discovery and inclusion of Python core regressions"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, - "claude_handoff": {"status": "pending"}, + "claude_full_gate": {"revision": "3555a91", "tests": 458, "seconds": 467.018, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "result": "pass"}, + "claude_provider_label_gate": {"tests": 63, "seconds": 70.391, "result": "pass", "scope": "Version-bound firstParty label normalization added after the 458-test gate; real saved native stream decodes to verified Sonnet 5 with auxiliary usage retained. The final managed candidate's mandatory full suite must include this small mapping patch."}, + "claude_independent_review": {"revision": "3555a91", "run_id": "b9501b55-1c65-4b32-a03f-1087cca8fefc", "model": "gpt-6.1-sol", "effort": "low", "verdict": "clean", "declared_checks": 3, "check_result": "all_passed", "terminal_state": "succeeded", "terminal_version": 23, "scope": "Argument framing and correlated native stream evidence; subsequent version-bound provider label patch is separately targeted and will be inspected in delivery review."}, + "claude_handoff": {"status": "pass", "run_id": "b9501b55-1c65-4b32-a03f-1087cca8fefc", "harness": "Claude Code CLI", "version": "2.1.220", "writer_model": "claude-sonnet-5", "calls": ["ToolSearch", "squad_status", "squad_handoff_claim", "squad_handoff_complete", "squad_status"], "terminal": {"state": "succeeded", "version": 23}, "permission_denials": 0, "stdout_sha256": "2b3aa171838e222249de7663c8d5ead499fc726975a516377bca32d7691308cb", "seconds": 38.61665333295241, "prior_attempts": ["Zero-tool probe ended with MCP still pending and did not invoke any tool", "ToolSearch-enabled probe claimed but falsely reported an artifact hash mismatch; deterministic exact comparison proved all refs equal. Its claim was retained and renewed before actual completion."], "usage": {"input_tokens": 12, "output_tokens": 2347, "cache_read_input_tokens": 77771, "cache_creation_input_tokens": 16166}, "scope": "Actual portable CLI/MCP handoff and acceptance, not Claude desktop local Code-tab UI proof"}, "grok_operation": {"status": "pass", "old_version_blocker": "HTTP 426: 0.2.111 rejected, requires 1.0.13 or later", "user_authorized_update": true, "updated_version": "1.0.46 (2765805b9442) [stable]", "old_executable_backup": "private_probe_directory", "native_model": "grok-4.7-build", "tool": "devsquad__squad_status", "observed": {"run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13}, "stdout_sha256": "bc031b922e8c36b3cdfff5c7c717916a27019a19d650f67cd7402b832544bfdd", "stderr_sha256": "1be68163f727c0712fefcd45b80d5f1d383ee995ff8f61f7da601fe82ee4a9b3", "usage": {"input_tokens": 34088, "cache_read_input_tokens": 69888, "output_tokens": 434, "reasoning_tokens": 285, "total_tokens": 104410, "turns": 3}, "extra_bounded_smoke": {"result": "pass", "stdout_sha256": "cbc772baa3dabf93b833cf9843c5a873bb72cbebf9fb8ddf7d9b5ae7f12a8f43", "total_tokens": 103523}, "scope": "native Grok Build MCP operation, not a verified implementation/reviewer adapter"}, "gemini_antigravity": {"status": "pass_cli_mcp", "version": "1.2.13", "native_model": "gemini-3.8-flash-low", "tool": "devsquad/squad_status", "permission": "existing project-scoped status grant, plan mode and sandbox", "schema_discovery": "read only DevSquad squad_status tool schema", "observed": {"run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13}, "seconds": 10.62698745902162, "stdout_sha256": "8781e538501b3553c100ca754d2880c8f45a011062eddd5fd3b108cf0878181d", "usage": {"input_tokens": 33060, "output_tokens": 220, "thinking_tokens": 0, "cache_read_tokens": 44852, "total_tokens": 33280}, "initial_probe": "Argument parser rejected unitless print timeout before generation; corrected to 120s", "ide_ui": "computer_use_permission_denied_do_not_bypass"}, "limitations": ["R4 catalog/quota, R5 public trials/outcomes, remaining R6 UX and R7 Council are separate open scope", "Jev runtime remains off and its one-request allowance is spent", "No paid API fallback, purchases, resets, global AI settings, push or external messages"] diff --git a/plugin/core/src/devsquad/claude_identity.py b/plugin/core/src/devsquad/claude_identity.py index 681a4dd..399956b 100644 --- a/plugin/core/src/devsquad/claude_identity.py +++ b/plugin/core/src/devsquad/claude_identity.py @@ -241,7 +241,12 @@ def observed_identity(native: Any, adapter: dict[str, Any], or profile.get("harness") != "claude" or profile.get("permission_policy") != "workspace_write"): raise ContractError("Claude identity does not match the frozen adapter") - if any(entry.get("provider", adapter["model_provider"]) != adapter["model_provider"] for entry in models.values()): + # Native 2.1.220 uses a transport label, not a model-provider ID. Retain + # the raw field and map only the label observed under this verified CLI. + allowed_providers = {adapter["model_provider"]} + if adapter["harness_version"] == "2.1.220 (Claude Code)": + allowed_providers.add("firstParty") + if any(entry.get("provider", adapter["model_provider"]) not in allowed_providers for entry in models.values()): raise ContractError("Claude reported provider contradicts the frozen adapter") return { "harness": "claude", "harness_version": adapter["harness_version"], diff --git a/test/core/test_claude_identity.py b/test/core/test_claude_identity.py index cd87f8d..796fdf3 100644 --- a/test/core/test_claude_identity.py +++ b/test/core/test_claude_identity.py @@ -272,6 +272,23 @@ def test_stream_session_model_proof_is_revalidated_on_import(self): with self.subTest(field=field, value=value), self.assertRaises(ContractError): validate_implementation_evidence(changed, snapshot) + def test_native_first_party_transport_label_is_version_bound_not_a_provider_override(self): + from devsquad.claude_identity import observed_identity + records = self.stream_records() + for entry in records[-1]["modelUsage"].values(): + entry["provider"] = "firstParty" + snapshot = self.snapshot() + observed = self.run_stream(records)["attempt"]["observed_identity"] + self.assertEqual(observed["model_provider"], "anthropic") + self.assertEqual(observed["native_evidence"]["model_usage"][self.MODEL]["provider"], "firstParty") + with self.assertRaisesRegex(ContractError, "provider"): + observed_identity(observed["native_evidence"], + {**snapshot["implementation_adapter"], "harness_version": "unverified-version"}, + snapshot["routing"]["roles"]["implementer"]["selected"]["profile"]) + records[-1]["modelUsage"]["claude-haiku-4-5-20251001"]["provider"] = "other-provider" + with self.assertRaisesRegex(ContractError, "implementation failed"): + self.run_stream(records) + def test_family_alias_resolves_only_to_reported_concrete_model(self): evidence = self.run_document(self.document(), requested_model="sonnet") attempt = evidence["attempt"] From 710e56a903f8f5e44fe8a3efcccd549ffd891c8b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 16:42:43 -0700 Subject: [PATCH 156/197] WIP checkpoint: Record second managed delivery failure and narrow native diagnostics (2026-10-01 16:42) --- docs/plans/engineering-team/RESUME.md | 19 +++++++++++++++++++ .../R8-installed-workflows-2026-10-01.json | 2 ++ 2 files changed, 21 insertions(+) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index ba58ef6..b424873 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -4,6 +4,25 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ## Current position — October 1, 2026 +### Active continuation — managed delivery failure remains unresolved + +HEAD `6f51db9` is installed as `0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef`, +with payload drift false and `pip check` passed. The second genuine G4 delivery, +`8c116b7d-3416-4f80-a990-d610e2a5a558`, failed at version 13 with one +implementer invocation and `native_result_invalid`; no candidate, reviewer or +mandatory checks were produced. Its 6,183 native output bytes have only a safe +digest/typed failure receipt, not raw logs. Preserve both failed runs. + +A private tool-free replay of the frozen adapter/profile and workspace passes +both with ordinary input and with the snapshot file at EOF inherited on fd 0. +Therefore input inheritance is **not a demonstrated cause**. A bounded one-Read +diagnostic is now capturing native output privately under +`/Users/Dikshant/.devsquad/private-probes/r8-installed-workflows-20261001`. +Next: inspect only event/identity/error metadata, repair the demonstrated +transport problem with regressions, refresh safely, then retry genuine G4 +Claude implementation → independent Codex review → mandatory tests under a +new run. Do not claim managed delivery passed or repeat completed host proofs. + ### Latest continuation — Claude launch/stream repair, Grok and Gemini proof The frozen stream/framing gate at `3555a91` passes **458 tests in 467.018s**, diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index 42ab9be..129c2be 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -18,6 +18,8 @@ "sdk_tests": {"tests": 22, "seconds": 5.585, "skips": 0, "result": "pass"} }, "delivery": {"status": "initial_attempt_failed_before_implementation", "run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13, "worker_invocations": 1, "reported_tokens": 0, "cause": "Claude variadic --tools consumes the final positional prompt without an explicit -- separator", "scope": "G4 exact-target check discovery and inclusion of Python core regressions"}, + "second_installation": {"source_checkpoint": "6f51db9", "release": "0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef", "source_digest": "7d7e408303b3f17043c0f413d2d9828feb390a73dacff0bce358921fe1802365", "payload_drift": false, "pip_check": "pass", "prior_releases_retained": true}, + "delivery_retry": {"run_id": "8c116b7d-3416-4f80-a990-d610e2a5a558", "state": "failed", "version": 13, "worker_invocations": 1, "reason": "native_result_invalid", "output_bytes": 6183, "output_sha256": "0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485", "candidate_produced": false, "review_and_checks": "not_started", "cause": "unresolved; tool-free frozen-context replays pass including inherited regular-file stdin at EOF"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, "claude_full_gate": {"revision": "3555a91", "tests": 458, "seconds": 467.018, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "result": "pass"}, "claude_provider_label_gate": {"tests": 63, "seconds": 70.391, "result": "pass", "scope": "Version-bound firstParty label normalization added after the 458-test gate; real saved native stream decodes to verified Sonnet 5 with auxiliary usage retained. The final managed candidate's mandatory full suite must include this small mapping patch."}, From 1781b89b1228dfca2ca9148df437af3ac04b971f Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 16:48:57 -0700 Subject: [PATCH 157/197] WIP checkpoint: Repair detached Claude saved-login environment and synthetic auth error classification (2026-10-01 16:48) --- docs/plans/engineering-team/CONTRACTS.md | 6 +++++ docs/plans/engineering-team/RESUME.md | 24 +++++++++++++------ .../R8-installed-workflows-2026-10-01.json | 1 + .../src/devsquad/claude_delivery_worker.py | 13 +++++++--- plugin/core/src/devsquad/claude_identity.py | 6 +++-- plugin/core/src/devsquad/service.py | 10 +++++++- test/core/test_claude_identity.py | 19 +++++++++++++++ test/core/test_service.py | 13 ++++++++++ 8 files changed, 79 insertions(+), 13 deletions(-) diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index af5ca58..e0bdd8f 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -100,6 +100,9 @@ again on coordinator import, before candidate finalization. The tested Claude 2.1.220 transport label `firstParty` maps to Anthropic only for that verified harness version; the original label remains in evidence. Unknown or contradictory provider labels still block identity verification. +Native error streams can report synthetic assistant messages; their strict +final error terminal is classified without claiming writer identity. A native +not-logged-in error reports AUTH_ERROR, ahead of a concurrent rate banner. New delivery review imports and acceptance require verified different reported model IDs, regardless of requested aliases, family labels or harness names. @@ -148,6 +151,9 @@ Execution completion is separate from deliverable validity and acceptance. Empty Use Git's tracked-file inventory and explicit task scope for context. Preserve filenames with spaces, TSX/JSX and other tracked extensions. Bound bytes and document exclusions; never silently truncate required evidence. Exclude runtime state, secrets and ignored files by default; an explicitly required ignored input needs an intentional input artifact. Large context should use native scoped filesystem access when supported rather than concatenating every file. Workers get `DEVSQUAD_WORKER=1`, run/attempt IDs and a delegation-depth guard. DevSquad's worker-facing MCP tools reject new team starts and workflow mutations, and legacy hooks honor the guard. Native authentication remains available through the provider's normal mechanism, but credentials and environment contents are not logged. Do not assume prompt text alone stops recursive delegation. +Detached launch preserves PATH, the frozen package PYTHONPATH, HOME and USER +for the normal saved-login mechanism. It does not inherit API credentials, +provider overrides or arbitrary host environment variables. ## 4. Durable state, concurrency and recovery diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index b424873..8d4d1ba 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -13,13 +13,23 @@ implementer invocation and `native_result_invalid`; no candidate, reviewer or mandatory checks were produced. Its 6,183 native output bytes have only a safe digest/typed failure receipt, not raw logs. Preserve both failed runs. -A private tool-free replay of the frozen adapter/profile and workspace passes -both with ordinary input and with the snapshot file at EOF inherited on fd 0. -Therefore input inheritance is **not a demonstrated cause**. A bounded one-Read -diagnostic is now capturing native output privately under -`/Users/Dikshant/.devsquad/private-probes/r8-installed-workflows-20261001`. -Next: inspect only event/identity/error metadata, repair the demonstrated -transport problem with regressions, refresh safely, then retry genuine G4 +A private tool-free replay passes even with the snapshot file at EOF on fd 0; +input inheritance is **not the cause**. One-Read replays pass under the operator +environment, but reproduce failure under the actual minimal detached +environment: a synthetic assistant model and a native `is_error: true`, +not-logged-in terminal, zero model usage. HOME alone still fails; HOME plus USER +restores the saved subscription login and passes the same one-Read replay. +No other environment variables or API credentials are needed. Raw output stays +private; redacted event/identity metadata is recorded in installed evidence. + +The repair preserves only HOME and USER alongside PATH/frozen PYTHONPATH at +the detached boundary. Native error terminals are classified before requiring +writer identity; a synthetic not-logged-in stream remains unverified and +reports AUTH_ERROR instead of generic CLI_ERROR, ahead of rate banners. Two red +regressions reproduced the omissions before source repair. The 50-test affected +gate passes in 69.606s; 227 Bash assertions and generated reference/diff pass. +The final auth-priority refinement passes a 22-test focused rerun in 3.652s. +Next: checkpoint after affected/Bash gates, refresh safely, then retry genuine G4 Claude implementation → independent Codex review → mandatory tests under a new run. Do not claim managed delivery passed or repeat completed host proofs. diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index 129c2be..58454df 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -20,6 +20,7 @@ "delivery": {"status": "initial_attempt_failed_before_implementation", "run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13, "worker_invocations": 1, "reported_tokens": 0, "cause": "Claude variadic --tools consumes the final positional prompt without an explicit -- separator", "scope": "G4 exact-target check discovery and inclusion of Python core regressions"}, "second_installation": {"source_checkpoint": "6f51db9", "release": "0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef", "source_digest": "7d7e408303b3f17043c0f413d2d9828feb390a73dacff0bce358921fe1802365", "payload_drift": false, "pip_check": "pass", "prior_releases_retained": true}, "delivery_retry": {"run_id": "8c116b7d-3416-4f80-a990-d610e2a5a558", "state": "failed", "version": 13, "worker_invocations": 1, "reason": "native_result_invalid", "output_bytes": 6183, "output_sha256": "0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485", "candidate_produced": false, "review_and_checks": "not_started", "cause": "unresolved; tool-free frozen-context replays pass including inherited regular-file stdin at EOF"}, + "detached_auth_diagnosis": {"result": "reproduced_and_source_repaired", "original_environment": "PATH and frozen PYTHONPATH, without HOME/USER", "native_failure": "Synthetic assistant model, is_error true and not-logged-in terminal; no model usage", "home_only": "still_failed", "home_plus_user": "pass_one_Read_with_verified_claude_sonnet_5", "failure_stream_sha256": "f7f7a53de76012c54c4a0757aa346124553366bf6b1f0c120cb7288ce1491ab7", "home_user_stream_sha256": "1061f1ca7688fa178b4add47934c5341287d02c66a4ba468d91343398fa830d3", "red_regressions": {"tests": 2, "errors": 2}, "repair": "Allowlist HOME and USER in detached launch; classify synthetic native error terminals without claiming writer identity. API keys and provider overrides remain excluded.", "focused_gate": {"tests": 50, "seconds": 69.606, "result": "pass"}, "auth_priority_rerun": {"tests": 22, "seconds": 3.652, "result": "pass"}, "verification": "source_targeted_pass; managed_delivery_still_pending"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, "claude_full_gate": {"revision": "3555a91", "tests": 458, "seconds": 467.018, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "result": "pass"}, "claude_provider_label_gate": {"tests": 63, "seconds": 70.391, "result": "pass", "scope": "Version-bound firstParty label normalization added after the 458-test gate; real saved native stream decodes to verified Sonnet 5 with auxiliary usage retained. The final managed candidate's mandatory full suite must include this small mapping patch."}, diff --git a/plugin/core/src/devsquad/claude_delivery_worker.py b/plugin/core/src/devsquad/claude_delivery_worker.py index 9318ec1..28d0e80 100644 --- a/plugin/core/src/devsquad/claude_delivery_worker.py +++ b/plugin/core/src/devsquad/claude_delivery_worker.py @@ -199,8 +199,6 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: ), None) if timed_out: error_code = "TIMEOUT" - elif completed.returncode != 0 and error_code is None: - error_code = "CLI_ERROR" provider_document = None try: from .claude_identity import decode_native_result @@ -212,10 +210,17 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: provider_text = str( safe_document.get("result") or safe_document.get("error") or "" ) - error_code = error_code or next(( + native_code = next(( code for code, pattern in adapter["error_patterns"].items() if re.search(pattern, provider_text, re.IGNORECASE) ), None) + if (safe_document.get("is_error") is True + and re.search(r"\bnot logged in\b", provider_text, re.IGNORECASE)): + native_code = "AUTH_ERROR" + if error_code != "TIMEOUT": + reported_codes = {error_code, native_code} + error_code = next((code for code in ("AUTH_ERROR", "RATE_LIMITED") + if code in reported_codes), error_code or native_code) if error_code is None and re.search( adapter["denied_pattern"], provider_text, re.IGNORECASE, ): @@ -223,6 +228,8 @@ def run(snapshot: dict[str, Any]) -> dict[str, Any]: raise ClaudeResultError(error_code or "CLI_ERROR", failure_diagnostics( stdout, provider_document, "native_result_invalid", )) from exc + if completed.returncode != 0 and error_code is None: + error_code = "CLI_ERROR" if error_code is not None: raise ClaudeResultError(error_code, failure_diagnostics( stdout, provider_document, "execution_failed", diff --git a/plugin/core/src/devsquad/claude_identity.py b/plugin/core/src/devsquad/claude_identity.py index 399956b..e0c60c0 100644 --- a/plugin/core/src/devsquad/claude_identity.py +++ b/plugin/core/src/devsquad/claude_identity.py @@ -155,6 +155,10 @@ def decode_native_result(payload: bytes | str) -> tuple[dict[str, Any], dict[str raise ContractError("Claude stream requires one final result") document = terminals[0] session = _session(document.get("session_id")) + # Native authentication failures emit a synthetic assistant model. + # Retain the error terminal for classification, never writer identity. + if document.get("is_error") is True: + return document, None models = [] for item in records: if item.get("type") != "assistant": @@ -165,8 +169,6 @@ def decode_native_result(payload: bytes | str) -> tuple[dict[str, Any], dict[str raise ContractError("Claude writer message is not session-correlated") models.append(_model(message.get("model"))) if not models: - if document.get("is_error") is True: - return document, None raise ContractError("Claude stream has no reported writer messages") return document, {"session_id": session, "models": sorted(set(models)), "message_count": len(models)} if not isinstance(document, dict): diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 1b26676..22f1750 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -1,6 +1,7 @@ """Durable M2 application operations shared by CLI and later MCP surfaces.""" from __future__ import annotations +import getpass import hashlib import json import os @@ -995,7 +996,14 @@ def start( def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, digest: str) -> int: command = [sys.executable, "-P", "-m", "devsquad.detached", "--database", str(self.database), "--artifacts", str(self.artifacts), "--run-id", run_id, "--expected-version", str(expected_version), "--package-digest", digest] - environment = {"PATH": os.environ.get("PATH", ""), "PYTHONPATH": str(package)} + # Claude's native saved-login lookup needs the login name and HOME. + # Keep this explicit: ambient API keys/provider overrides never cross + # the detached boundary, and package imports remain frozen. + environment = { + "PATH": os.environ.get("PATH", ""), "PYTHONPATH": str(package), + "HOME": str(Path.home()), + "USER": os.environ.get("USER") or getpass.getuser(), + } log_dir=self.runtime/"private-logs"; log_dir.mkdir(parents=True,exist_ok=True) with (log_dir/f"{run_id}.supervisor.log").open("ab",buffering=0) as diagnostic: process = subprocess.Popen(command, cwd=self.runtime, env=environment, stdin=subprocess.DEVNULL, stdout=diagnostic, stderr=diagnostic, start_new_session=True, close_fds=True) diff --git a/test/core/test_claude_identity.py b/test/core/test_claude_identity.py index 796fdf3..f6cfb77 100644 --- a/test/core/test_claude_identity.py +++ b/test/core/test_claude_identity.py @@ -371,6 +371,25 @@ def test_only_a_success_result_envelope_can_supply_identity(self): with self.assertRaises(ContractError): self.run_document(document) + def test_synthetic_auth_failure_is_classified_without_verifying_identity(self): + from devsquad.claude_identity import ClaudeResultError, decode_native_result + records = self.stream_records() + records[0]["message"]["model"] = "" + records[-1].update({"is_error": True, "result": "Not logged in. Please run /login.", "modelUsage": {}}) + payload = "\n".join(json.dumps(item) for item in records) + document, messages = decode_native_result(payload) + self.assertIsNone(messages) + self.assertTrue(document["is_error"]) + original_binary = self.binary.read_text() + for stderr in ("", "rate limit"): + with self.subTest(stderr=stderr): + self.binary.write_text(original_binary + f"printf '%s\\n' {shlex.quote(stderr)} >&2\nexit 1\n") + with self.assertRaises(ClaudeResultError) as raised: + self.run_stream(records) + self.assertEqual(raised.exception.code, "AUTH_ERROR") + self.assertEqual(raised.exception.diagnostics["identity_status"], "unverified") + self.assertEqual(raised.exception.diagnostics["model_usage"], {}) + def test_missing_or_invalid_native_session_cannot_supply_identity(self): for session in (None, "", " ", 42, [], "s" * 10000): with self.subTest(session=session): diff --git a/test/core/test_service.py b/test/core/test_service.py index 4f70380..c3ebb0c 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -86,6 +86,19 @@ def request_then_crash(cancel_run_id, attempt_token): class ServiceTest(unittest.TestCase): + def test_detached_environment_preserves_home_without_ambient_api_credentials(self): + fake_home = self.root / "signed-in-home" + with mock.patch.dict(os.environ, { + "HOME": str(fake_home), "USER": "signed-in-fixture", "ANTHROPIC_API_KEY": "must-not-pass", + "OPENAI_API_KEY": "must-not-pass", "BASH_ENV": "must-not-pass", + }), mock.patch("devsquad.service.subprocess.Popen") as popen: + popen.return_value.pid = 12345 + self.service._spawn_daemon("private-fixture", 1, self.root / "package", "a" * 64) + environment = popen.call_args.kwargs["env"] + self.assertEqual(environment["HOME"], str(fake_home)) + self.assertEqual(environment["USER"], "signed-in-fixture") + self.assertEqual(set(environment), {"HOME", "USER", "PATH", "PYTHONPATH"}) + def setUp(self): self.temp = tempfile.TemporaryDirectory(prefix="devsquad-service-") self.root = Path(self.temp.name); self.repo = self.root / "repo"; self.runtime = self.root / "runtime" From 4a6f91b500464235f735e60d390a582c8b066f69 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 16:51:01 -0700 Subject: [PATCH 158/197] WIP checkpoint: Record auth-repaired installation and active genuine delivery (2026-10-01 16:51) --- docs/plans/engineering-team/RESUME.md | 12 ++++++++++++ .../evidence/R8-installed-workflows-2026-10-01.json | 2 ++ 2 files changed, 14 insertions(+) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 8d4d1ba..8f63e35 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,18 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Active continuation — managed delivery failure remains unresolved +The auth repair is committed at `1781b89` and installed as +`0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef`; payload drift is false, +`pip check` passes, and previous releases remain recoverable. A fresh genuine +G4 delivery is running: **`45667697-1aa2-4a49-9737-2c4e637ebd26`**, initially +running/version 4. It freezes the repaired source, scoped Claude Sonnet writer, +independent Codex gpt-6.1-sol/low adversarial review, Bash check and mandatory +spawn-safe full Python core runner (900s check timeout). Do not start a second +full core gate or edit its frozen candidate. Inspect saved status/evidence, +finish the fenced host disposition if eligible, then integrate only accepted +source changes and refresh the final installation. Managed delivery is not +yet a pass. The older installation/failures below are retained history. + HEAD `6f51db9` is installed as `0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef`, with payload drift false and `pip check` passed. The second genuine G4 delivery, `8c116b7d-3416-4f80-a990-d610e2a5a558`, failed at version 13 with one diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index 58454df..e9dd9c4 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -19,6 +19,8 @@ }, "delivery": {"status": "initial_attempt_failed_before_implementation", "run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13, "worker_invocations": 1, "reported_tokens": 0, "cause": "Claude variadic --tools consumes the final positional prompt without an explicit -- separator", "scope": "G4 exact-target check discovery and inclusion of Python core regressions"}, "second_installation": {"source_checkpoint": "6f51db9", "release": "0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef", "source_digest": "7d7e408303b3f17043c0f413d2d9828feb390a73dacff0bce358921fe1802365", "payload_drift": false, "pip_check": "pass", "prior_releases_retained": true}, + "auth_repaired_installation": {"source_checkpoint": "1781b89", "release": "0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef", "source_digest": "68d542f6e8ea69c9bdefdfd4625af45ea46a09fb7926964397cbad3d96c83f28", "payload_drift": false, "pip_check": "pass", "previous_releases_retained": true}, + "auth_repaired_delivery": {"run_id": "45667697-1aa2-4a49-9737-2c4e637ebd26", "status": "running", "candidate_produced": false, "review_and_checks": "pending", "scope": "Genuine installed Claude implementation of G4, independent Codex review and mandatory Bash/full Python checks"}, "delivery_retry": {"run_id": "8c116b7d-3416-4f80-a990-d610e2a5a558", "state": "failed", "version": 13, "worker_invocations": 1, "reason": "native_result_invalid", "output_bytes": 6183, "output_sha256": "0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485", "candidate_produced": false, "review_and_checks": "not_started", "cause": "unresolved; tool-free frozen-context replays pass including inherited regular-file stdin at EOF"}, "detached_auth_diagnosis": {"result": "reproduced_and_source_repaired", "original_environment": "PATH and frozen PYTHONPATH, without HOME/USER", "native_failure": "Synthetic assistant model, is_error true and not-logged-in terminal; no model usage", "home_only": "still_failed", "home_plus_user": "pass_one_Read_with_verified_claude_sonnet_5", "failure_stream_sha256": "f7f7a53de76012c54c4a0757aa346124553366bf6b1f0c120cb7288ce1491ab7", "home_user_stream_sha256": "1061f1ca7688fa178b4add47934c5341287d02c66a4ba468d91343398fa830d3", "red_regressions": {"tests": 2, "errors": 2}, "repair": "Allowlist HOME and USER in detached launch; classify synthetic native error terminals without claiming writer identity. API keys and provider overrides remain excluded.", "focused_gate": {"tests": 50, "seconds": 69.606, "result": "pass"}, "auth_priority_rerun": {"tests": 22, "seconds": 3.652, "result": "pass"}, "verification": "source_targeted_pass; managed_delivery_still_pending"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, From 3b67fff2b3993ebec30d721b4eb92e354085f64f Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 17:32:45 -0700 Subject: [PATCH 159/197] WIP checkpoint: Record verified Claude candidate, dead-review cancellation and corrective delivery (2026-10-01 17:32) --- docs/plans/engineering-team/RESUME.md | 23 ++++++++++++++++++- docs/plans/engineering-team/backlog.json | 4 ++-- .../R8-installed-workflows-2026-10-01.json | 3 ++- 3 files changed, 26 insertions(+), 4 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 8f63e35..2a28133 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,10 +6,31 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Active continuation — managed delivery failure remains unresolved +**Latest active run: `288ee6f9-f503-421a-b81a-d50ee0cf44d8`.** The prior run +`45667697…` produced a verified Claude Sonnet 5 implementation and frozen +candidate `5cb9eab81bf4f7ba01861ffb05193a8a67521200`, but its reviewer runner +and child died without an exit receipt. The exact cause is unknown; stdout, +stderr and supervisor log were empty. Both process identities are dead and +public orphan cancellation terminalized it **cancelled/version 17**. Preserve +that truthful receipt and candidate; no review/test pass is inferred. + +The operator's isolated real-Git assessment found that `_has_tracked_files` +wrongly selects Python discovery for README-only test directories and a +`tests` symlink. The new bounded run starts from the retained candidate, repairs +these cases with Claude, and asks Codex to inspect the **whole** G4 diff from +`1781b89` to the new candidate. The review focus explicitly forbids running +tests/nested supervisors; the coordinator owns the mandatory Bash/full Python +checks. Codex reports ordinary usage allowed, not a current quota block. +Installed SDK transport rerun passes **9 tests in 3.191s**. Do not duplicate a +full core gate or modify frozen runtime packages/worktrees while this runs. +Next: inspect this run's final review/check receipts, accept only if all gates +and the reproduced cases pass, integrate the full accepted G4 diff, checkpoint +and refresh the final installation; then recheck Gemini CLI/MCP once. + The auth repair is committed at `1781b89` and installed as `0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef`; payload drift is false, `pip check` passes, and previous releases remain recoverable. A fresh genuine -G4 delivery is running: **`45667697-1aa2-4a49-9737-2c4e637ebd26`**, initially +G4 delivery was launched: **`45667697-1aa2-4a49-9737-2c4e637ebd26`**, initially running/version 4. It freezes the repaired source, scoped Claude Sonnet writer, independent Codex gpt-6.1-sol/low adversarial review, Bash check and mandatory spawn-safe full Python core runner (900s check timeout). Do not start a second diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 8a0b15f..7ae5e17 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -21,8 +21,8 @@ "status": "partial", "artifact": "evidence/R3b2-eligibility-partial-2026-10-01.json", "scope": "Reader full gate passed at a4a87fd; schema-15 explicit revisions and shared lifecycle eligibility implemented with focused public fixtures; no install or provider calls", - "next_action": "Full spawn-safe integration at dc68944 passes 455 tests (two optional-SDK skips, zero errors/failures/unraisable diagnostics) after the lifecycle clock fixture repair. Bounded upgrade review is clean. Safely refresh using the existing Python 3.12/MCP wheelhouse, verify installed SDK, then prove Claude delivery/handoff, Grok and updated Gemini/Antigravity.", - "limitations": "R4 catalog/quota, R5 public controller/outcomes, R6 UX and R7 Council open. Installed schema 13 has no nonterminal runs but is not yet refreshed. Claude/Grok/updated Gemini live proofs pending." + "next_action": "Installed schema-15 runtime is refreshed through the tested Claude framing/identity and detached auth repairs. Real Claude host handoff, Grok operation and Gemini CLI/MCP pass. Genuine managed Claude implementation succeeded, but the reviewer died without a receipt. Continue the saved G4 correction run 288ee6f9-f503-421a-b81a-d50ee0cf44d8 through independent review, mandatory tests and fenced acceptance, then integrate and refresh.", + "limitations": "R4 catalog/quota, R5 public controller/outcomes, remaining R6 UX and R7 Council remain open. Managed implementation/review/tests acceptance is pending. Gemini CLI/MCP works; IDE UI permission is denied. Jev remains off; no paid API fallback or reset use authorized." }, "planning_checkpoint": { "recorded_on": "2026-10-01", diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index e9dd9c4..2aea793 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -20,7 +20,8 @@ "delivery": {"status": "initial_attempt_failed_before_implementation", "run_id": "8a8a4548-13c1-4a2f-9190-870112f63ef2", "state": "failed", "version": 13, "worker_invocations": 1, "reported_tokens": 0, "cause": "Claude variadic --tools consumes the final positional prompt without an explicit -- separator", "scope": "G4 exact-target check discovery and inclusion of Python core regressions"}, "second_installation": {"source_checkpoint": "6f51db9", "release": "0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef", "source_digest": "7d7e408303b3f17043c0f413d2d9828feb390a73dacff0bce358921fe1802365", "payload_drift": false, "pip_check": "pass", "prior_releases_retained": true}, "auth_repaired_installation": {"source_checkpoint": "1781b89", "release": "0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef", "source_digest": "68d542f6e8ea69c9bdefdfd4625af45ea46a09fb7926964397cbad3d96c83f28", "payload_drift": false, "pip_check": "pass", "previous_releases_retained": true}, - "auth_repaired_delivery": {"run_id": "45667697-1aa2-4a49-9737-2c4e637ebd26", "status": "running", "candidate_produced": false, "review_and_checks": "pending", "scope": "Genuine installed Claude implementation of G4, independent Codex review and mandatory Bash/full Python checks"}, + "auth_repaired_delivery": {"run_id": "45667697-1aa2-4a49-9737-2c4e637ebd26", "status": "cancelled", "version": 17, "candidate_produced": true, "candidate_commit": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_patch_sha256": "bb47b4b5522826609cabc335e0984b99ffd8311ad620a4e7cfe04c05a6906e26", "writer": {"model": "claude-sonnet-5", "verification": "verified", "model_source": "claude.stream.assistant.message.model", "reported_usage": {"input_tokens": 36, "output_tokens": 29404, "total_tokens": 29440}}, "review_and_checks": "no_completed_evidence", "reviewer_failure": "runner and child died without an exit receipt; exact cause unknown; zero capture/log bytes; dead ownership safely cancelled", "scope": "Genuine installed Claude implementation passed, but independent review/test acceptance did not complete"}, + "candidate_correction": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "status": "running", "base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "operator_red_cases": ["README-only tests tree incorrectly selects Python discovery", "tests symlink incorrectly selects Python discovery"], "review_scope": "Complete G4 diff from original base, not only the corrective delta", "mandatory_checks": "Bash plus spawn-safe full Python core runner, coordinator-owned"}, "delivery_retry": {"run_id": "8c116b7d-3416-4f80-a990-d610e2a5a558", "state": "failed", "version": 13, "worker_invocations": 1, "reason": "native_result_invalid", "output_bytes": 6183, "output_sha256": "0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485", "candidate_produced": false, "review_and_checks": "not_started", "cause": "unresolved; tool-free frozen-context replays pass including inherited regular-file stdin at EOF"}, "detached_auth_diagnosis": {"result": "reproduced_and_source_repaired", "original_environment": "PATH and frozen PYTHONPATH, without HOME/USER", "native_failure": "Synthetic assistant model, is_error true and not-logged-in terminal; no model usage", "home_only": "still_failed", "home_plus_user": "pass_one_Read_with_verified_claude_sonnet_5", "failure_stream_sha256": "f7f7a53de76012c54c4a0757aa346124553366bf6b1f0c120cb7288ce1491ab7", "home_user_stream_sha256": "1061f1ca7688fa178b4add47934c5341287d02c66a4ba468d91343398fa830d3", "red_regressions": {"tests": 2, "errors": 2}, "repair": "Allowlist HOME and USER in detached launch; classify synthetic native error terminals without claiming writer identity. API keys and provider overrides remain excluded.", "focused_gate": {"tests": 50, "seconds": 69.606, "result": "pass"}, "auth_priority_rerun": {"tests": 22, "seconds": 3.652, "result": "pass"}, "verification": "source_targeted_pass; managed_delivery_still_pending"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, From 8129b923982f328b3ce9d240570a325bbc022d87 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 17:38:21 -0700 Subject: [PATCH 160/197] WIP checkpoint: Record corrective candidate scope and Unicode discovery finding before full gate (2026-10-01 17:38) --- docs/plans/engineering-team/RESUME.md | 15 +++++++++++++++ .../R8-installed-workflows-2026-10-01.json | 2 +- 2 files changed, 16 insertions(+), 1 deletion(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 2a28133..63806d6 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -27,6 +27,21 @@ Next: inspect this run's final review/check receipts, accept only if all gates and the reproduced cases pass, integrate the full accepted G4 diff, checkpoint and refresh the final installation; then recheck Gemini CLI/MCP once. +The corrective Claude writer finished with candidate +`6375955f66ce87a541fdfbf60db6eb0fcd30a55e`; both README/symlink red cases now +pass. The reviewer/check worker is running version 13, with the coordinator's +Python regression runner observed active. **Actual frozen review base is +`5cb9eab`, not `1781b89`**: delivery preparation uses its implementation +baseline even though the requested task base/focus names the original base. +Do not claim whole-combined-diff review from that receipt alone; supplement it +with a real branch review `1781b89` → final candidate before integration. +An additional real-Git fixture reproduces missed Unicode `test_π.py` because +the current parser reads quoted newline `ls-tree` output. After this gate, +use a fenced host **revise**, request NUL-safe regular-blob filename discovery +plus Unicode/tab/newline positive regressions, and finish the revised candidate +through independent review and mandatory tests. Do not accept the known issue +or duplicate the running full gate. No frozen runtime file has been modified. + The auth repair is committed at `1781b89` and installed as `0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef`; payload drift is false, `pip check` passes, and previous releases remain recoverable. A fresh genuine diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index 2aea793..077a0c4 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -21,7 +21,7 @@ "second_installation": {"source_checkpoint": "6f51db9", "release": "0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef", "source_digest": "7d7e408303b3f17043c0f413d2d9828feb390a73dacff0bce358921fe1802365", "payload_drift": false, "pip_check": "pass", "prior_releases_retained": true}, "auth_repaired_installation": {"source_checkpoint": "1781b89", "release": "0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef", "source_digest": "68d542f6e8ea69c9bdefdfd4625af45ea46a09fb7926964397cbad3d96c83f28", "payload_drift": false, "pip_check": "pass", "previous_releases_retained": true}, "auth_repaired_delivery": {"run_id": "45667697-1aa2-4a49-9737-2c4e637ebd26", "status": "cancelled", "version": 17, "candidate_produced": true, "candidate_commit": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_patch_sha256": "bb47b4b5522826609cabc335e0984b99ffd8311ad620a4e7cfe04c05a6906e26", "writer": {"model": "claude-sonnet-5", "verification": "verified", "model_source": "claude.stream.assistant.message.model", "reported_usage": {"input_tokens": 36, "output_tokens": 29404, "total_tokens": 29440}}, "review_and_checks": "no_completed_evidence", "reviewer_failure": "runner and child died without an exit receipt; exact cause unknown; zero capture/log bytes; dead ownership safely cancelled", "scope": "Genuine installed Claude implementation passed, but independent review/test acceptance did not complete"}, - "candidate_correction": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "status": "running", "base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "operator_red_cases": ["README-only tests tree incorrectly selects Python discovery", "tests symlink incorrectly selects Python discovery"], "review_scope": "Complete G4 diff from original base, not only the corrective delta", "mandatory_checks": "Bash plus spawn-safe full Python core runner, coordinator-owned"}, + "candidate_correction": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "status": "running", "requested_base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_commit": "6375955f66ce87a541fdfbf60db6eb0fcd30a55e", "candidate_sha256": "77d16f59fcf8594615b25f774b3f106f4883565ff83a613453469c045be91b8a", "operator_red_cases": ["README-only tests tree incorrectly selects Python discovery", "tests symlink incorrectly selects Python discovery"], "operator_recheck": "both pass on corrective candidate", "remaining_operator_finding": "Git-quoted Unicode test filename test_π.py is missed; reproduced offline, requires fenced revision", "actual_review_base": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "review_scope_note": "Frozen delivery review uses implementer baseline, not original requested task base; full combined branch review remains required", "mandatory_checks": "Bash plus spawn-safe full Python core runner, coordinator-owned"}, "delivery_retry": {"run_id": "8c116b7d-3416-4f80-a990-d610e2a5a558", "state": "failed", "version": 13, "worker_invocations": 1, "reason": "native_result_invalid", "output_bytes": 6183, "output_sha256": "0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485", "candidate_produced": false, "review_and_checks": "not_started", "cause": "unresolved; tool-free frozen-context replays pass including inherited regular-file stdin at EOF"}, "detached_auth_diagnosis": {"result": "reproduced_and_source_repaired", "original_environment": "PATH and frozen PYTHONPATH, without HOME/USER", "native_failure": "Synthetic assistant model, is_error true and not-logged-in terminal; no model usage", "home_only": "still_failed", "home_plus_user": "pass_one_Read_with_verified_claude_sonnet_5", "failure_stream_sha256": "f7f7a53de76012c54c4a0757aa346124553366bf6b1f0c120cb7288ce1491ab7", "home_user_stream_sha256": "1061f1ca7688fa178b4add47934c5341287d02c66a4ba468d91343398fa830d3", "red_regressions": {"tests": 2, "errors": 2}, "repair": "Allowlist HOME and USER in detached launch; classify synthetic native error terminals without claiming writer identity. API keys and provider overrides remain excluded.", "focused_gate": {"tests": 50, "seconds": 69.606, "result": "pass"}, "auth_priority_rerun": {"tests": 22, "seconds": 3.652, "result": "pass"}, "verification": "source_targeted_pass; managed_delivery_still_pending"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, From e8bd9e5538ba42965f0ab2b4d132646c13ba49c3 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 17:47:52 -0700 Subject: [PATCH 161/197] WIP checkpoint: Preserve failed 471-test gate and launch bounded filename/build-helper repairs (2026-10-01 17:47) --- docs/plans/engineering-team/RESUME.md | 33 +++++++++++++++++++ .../R8-installed-workflows-2026-10-01.json | 2 ++ 2 files changed, 35 insertions(+) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 63806d6..21c2283 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,39 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Active continuation — managed delivery failure remains unresolved +**Current active run: `33c64397-3a57-42e7-82d5-773d50f466ee`**, launched via +the installed stable `squad fix`, original base `1781b89`, retained target +`6375955f66ce87a541fdfbf60db6eb0fcd30a55e`. Claude is repairing NUL-safe +Unicode/tab/newline test filenames and three `InstalledWheel*.build_python` +helpers in `test_cli.py`, `test_handoff_store.py`, `test_mcp.py`: unavailable +HOME-relative interpreter candidates must catch OSError and try the next +candidate. This new task freezes all five allowed files; the previous run's +immutable scope/checks were not widened. The full check explicitly sets +`DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3.12` +so wheel gates run, not silently skip. Keep check HOME isolation and identity +validation strict. No API fallback/settings/reset changes. + +The prior corrective run `288ee6f9…` completed a verified native Codex clean +review, explicitly reporting inspection of the complete G4 diff from 1781b89, +although its frozen comparison remains 5cb9eab → 6375955. Bash/diff checks +passed with verified candidate integrity. Full core gate: **471 tests, +473.762s, 3 errors, zero failures/unraisable, two optional-SDK skips**. +All three errors are the missing build-interpreter candidate under isolated +HOME; two were reproduced immediately by a focused offline check. No pass is +inferred. The operator used exact artifact-bound host **reject**, terminal +**failed/version 31**, retaining all receipts and the candidate. + +Next: finish current Claude → native Codex → mandatory tests; verify the +previous negative and filename positive fixtures; accept only a passing saved +candidate. Run one bounded saved full-combined branch review from 1781b89 to +the final candidate (no duplicate full gate), then integrate that full diff via +apply_patch, affected/Bash gates, checkpoint, final safe install and Gemini +CLI/MCP recheck. Current installation remains `68d542f6e8ea`, source HEAD is +clean except this recovery/evidence update. Private host helper is +`/Users/Dikshant/.devsquad/private-probes/r8-installed-workflows-20261001/complete-handoff.py`; +it revalidates artifact hashes and does not print claims. Do not reuse the old +native Claude handoff's prior claim for a new run. + **Latest active run: `288ee6f9-f503-421a-b81a-d50ee0cf44d8`.** The prior run `45667697…` produced a verified Claude Sonnet 5 implementation and frozen candidate `5cb9eab81bf4f7ba01861ffb05193a8a67521200`, but its reviewer runner diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index 077a0c4..ea47625 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -22,6 +22,8 @@ "auth_repaired_installation": {"source_checkpoint": "1781b89", "release": "0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef", "source_digest": "68d542f6e8ea69c9bdefdfd4625af45ea46a09fb7926964397cbad3d96c83f28", "payload_drift": false, "pip_check": "pass", "previous_releases_retained": true}, "auth_repaired_delivery": {"run_id": "45667697-1aa2-4a49-9737-2c4e637ebd26", "status": "cancelled", "version": 17, "candidate_produced": true, "candidate_commit": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_patch_sha256": "bb47b4b5522826609cabc335e0984b99ffd8311ad620a4e7cfe04c05a6906e26", "writer": {"model": "claude-sonnet-5", "verification": "verified", "model_source": "claude.stream.assistant.message.model", "reported_usage": {"input_tokens": 36, "output_tokens": 29404, "total_tokens": 29440}}, "review_and_checks": "no_completed_evidence", "reviewer_failure": "runner and child died without an exit receipt; exact cause unknown; zero capture/log bytes; dead ownership safely cancelled", "scope": "Genuine installed Claude implementation passed, but independent review/test acceptance did not complete"}, "candidate_correction": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "status": "running", "requested_base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_commit": "6375955f66ce87a541fdfbf60db6eb0fcd30a55e", "candidate_sha256": "77d16f59fcf8594615b25f774b3f106f4883565ff83a613453469c045be91b8a", "operator_red_cases": ["README-only tests tree incorrectly selects Python discovery", "tests symlink incorrectly selects Python discovery"], "operator_recheck": "both pass on corrective candidate", "remaining_operator_finding": "Git-quoted Unicode test filename test_π.py is missed; reproduced offline, requires fenced revision", "actual_review_base": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "review_scope_note": "Frozen delivery review uses implementer baseline, not original requested task base; full combined branch review remains required", "mandatory_checks": "Bash plus spawn-safe full Python core runner, coordinator-owned"}, + "corrective_gate": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "review": "verified native gpt-6.1-sol/low, clean; summary explicitly says full G4 diff inspected, frozen comparison 5cb9eab to 6375955", "bash_diff": "passed_with_verified_integrity", "full_core": {"tests": 471, "seconds": 473.7623341669996, "utc_seconds": 473.758578, "errors": 3, "failures": 0, "skips": 2, "unraisable": [], "result": "failed"}, "cause": "Three installed-wheel build_python helpers raise FileNotFoundError for nonexistent HOME-relative cached interpreter under deliberately isolated HOME instead of probing next candidate", "focused_reproduction": {"tests": 2, "errors": 2}, "host_disposition": "reject", "terminal": {"state": "failed", "version": 31}}, + "final_runtime_repair": {"run_id": "33c64397-3a57-42e7-82d5-773d50f466ee", "status": "running", "target": "6375955f66ce87a541fdfbf60db6eb0fcd30a55e", "scope": "NUL-safe filename discovery plus three unavailable-interpreter wheel-test helpers and regressions", "required_full_gate": "Explicit existing offline DEVSQUAD_BUILD_PYTHON, isolated HOME and strict ResourceWarning; no installation-test skip substitution"}, "delivery_retry": {"run_id": "8c116b7d-3416-4f80-a990-d610e2a5a558", "state": "failed", "version": 13, "worker_invocations": 1, "reason": "native_result_invalid", "output_bytes": 6183, "output_sha256": "0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485", "candidate_produced": false, "review_and_checks": "not_started", "cause": "unresolved; tool-free frozen-context replays pass including inherited regular-file stdin at EOF"}, "detached_auth_diagnosis": {"result": "reproduced_and_source_repaired", "original_environment": "PATH and frozen PYTHONPATH, without HOME/USER", "native_failure": "Synthetic assistant model, is_error true and not-logged-in terminal; no model usage", "home_only": "still_failed", "home_plus_user": "pass_one_Read_with_verified_claude_sonnet_5", "failure_stream_sha256": "f7f7a53de76012c54c4a0757aa346124553366bf6b1f0c120cb7288ce1491ab7", "home_user_stream_sha256": "1061f1ca7688fa178b4add47934c5341287d02c66a4ba468d91343398fa830d3", "red_regressions": {"tests": 2, "errors": 2}, "repair": "Allowlist HOME and USER in detached launch; classify synthetic native error terminals without claiming writer identity. API keys and provider overrides remain excluded.", "focused_gate": {"tests": 50, "seconds": 69.606, "result": "pass"}, "auth_priority_rerun": {"tests": 22, "seconds": 3.652, "result": "pass"}, "verification": "source_targeted_pass; managed_delivery_still_pending"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, From f1f0c942541c3ffc4de557218cf058728d19cc0b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 17:57:52 -0700 Subject: [PATCH 162/197] WIP checkpoint: Save final frozen candidate and exact combined-review next action (2026-10-01 17:57) --- docs/plans/engineering-team/RESUME.md | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 21c2283..9161bcb 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -18,6 +18,16 @@ immutable scope/checks were not widened. The full check explicitly sets so wheel gates run, not silently skip. Keep check HOME isolation and identity validation strict. No API fallback/settings/reset changes. +The writer finished candidate **`af994e948eb38764884d27fd61665efa776cb47b`**, +SHA-256 `efb502a38972a97aba89cce371b398e2ac7e7dd1b6dd9b0a8fa83a51d98d9f5c`, +five changed paths exactly as authorized. Unicode/README/symlink operator +fixtures now all pass. Native reviewer/check worker is running version 13. +The final full-combined review can use a prelaunch public branch-review task +with read scope limited to those five files, exact base 1781b89 → final target, +mandatory Bash plus affected tests with explicit offline build interpreter, +and no duplicate full gate. Do not broaden a frozen task or claim finished +checks before the saved packet exists. + The prior corrective run `288ee6f9…` completed a verified native Codex clean review, explicitly reporting inspection of the complete G4 diff from 1781b89, although its frozen comparison remains 5cb9eab → 6375955. Bash/diff checks From 6e6dcdfd1be338f3ec64e62e59dafff243e27258 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 18:03:29 -0700 Subject: [PATCH 163/197] WIP checkpoint: Record passing real 475-test delivery gate and active combined baseline review (2026-10-01 18:03) --- docs/plans/engineering-team/RESUME.md | 25 +++++++++++++++++++ .../R8-installed-workflows-2026-10-01.json | 5 ++-- 2 files changed, 28 insertions(+), 2 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 9161bcb..4eb6d82 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,31 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Active continuation — managed delivery failure remains unresolved +### Passing managed gate — final combined review / integration remains + +Current active full-combined branch review: **`ba14b743-9473-46fe-9e1f-d5b79fb032d4`**, +exact base `1781b89` → target **`af994e948eb38764884d27fd61665efa776cb47b`**, +read scope limited to the five changed files. Its diff/Bash/affected-wheel and +task-entry checks are mandatory; no duplicate full gate is declared. + +Managed delivery **`33c64397-3a57-42e7-82d5-773d50f466ee`** is now awaiting +host/version 23 with **accept_allowed true**. Actual Claude Sonnet 5 writer +and independent native gpt-6.1-sol/low reviewer are verified, review clean. +All three checks pass with unchanged/verified candidate integrity: diff, +227 Bash assertions, **475 core tests in 460.153s**, two optional-SDK skips, +zero errors/failures/unraisable diagnostics (UTC/monotonic ~460.516s agree). +The explicit build interpreter caused installed-wheel gates to run. All +operator README/symlink/Unicode cases pass. This is a genuine model workflow, +not fixture substitution. It is not terminal until fenced host acceptance. + +Next: inspect combined review's exact baseline/target and all required checks; +accept both saved runs only if clean/passing, then apply the prepared full G4 +diff from 1781b89 to af994 (five files) to the main project checkout, verify +blob hashes against the accepted candidate, affected/Bash gates and checkpoint. +Refresh the stable installation safely, verify drift/idempotence/doctor/SDK, +then one Gemini CLI/MCP status recheck against the final installed launcher. +Earlier failures below remain history; broader R4–R7 remain open. + **Current active run: `33c64397-3a57-42e7-82d5-773d50f466ee`**, launched via the installed stable `squad fix`, original base `1781b89`, retained target `6375955f66ce87a541fdfbf60db6eb0fcd30a55e`. Claude is repairing NUL-safe diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index ea47625..cd609e2 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "status": "claude_handoff_grok_gemini_pass_managed_delivery_pending", + "status": "managed_gates_pass_final_combined_review_and_install_pending", "source_checkpoint": "c7ccc02", "installation": { "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", @@ -23,7 +23,8 @@ "auth_repaired_delivery": {"run_id": "45667697-1aa2-4a49-9737-2c4e637ebd26", "status": "cancelled", "version": 17, "candidate_produced": true, "candidate_commit": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_patch_sha256": "bb47b4b5522826609cabc335e0984b99ffd8311ad620a4e7cfe04c05a6906e26", "writer": {"model": "claude-sonnet-5", "verification": "verified", "model_source": "claude.stream.assistant.message.model", "reported_usage": {"input_tokens": 36, "output_tokens": 29404, "total_tokens": 29440}}, "review_and_checks": "no_completed_evidence", "reviewer_failure": "runner and child died without an exit receipt; exact cause unknown; zero capture/log bytes; dead ownership safely cancelled", "scope": "Genuine installed Claude implementation passed, but independent review/test acceptance did not complete"}, "candidate_correction": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "status": "running", "requested_base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_commit": "6375955f66ce87a541fdfbf60db6eb0fcd30a55e", "candidate_sha256": "77d16f59fcf8594615b25f774b3f106f4883565ff83a613453469c045be91b8a", "operator_red_cases": ["README-only tests tree incorrectly selects Python discovery", "tests symlink incorrectly selects Python discovery"], "operator_recheck": "both pass on corrective candidate", "remaining_operator_finding": "Git-quoted Unicode test filename test_π.py is missed; reproduced offline, requires fenced revision", "actual_review_base": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "review_scope_note": "Frozen delivery review uses implementer baseline, not original requested task base; full combined branch review remains required", "mandatory_checks": "Bash plus spawn-safe full Python core runner, coordinator-owned"}, "corrective_gate": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "review": "verified native gpt-6.1-sol/low, clean; summary explicitly says full G4 diff inspected, frozen comparison 5cb9eab to 6375955", "bash_diff": "passed_with_verified_integrity", "full_core": {"tests": 471, "seconds": 473.7623341669996, "utc_seconds": 473.758578, "errors": 3, "failures": 0, "skips": 2, "unraisable": [], "result": "failed"}, "cause": "Three installed-wheel build_python helpers raise FileNotFoundError for nonexistent HOME-relative cached interpreter under deliberately isolated HOME instead of probing next candidate", "focused_reproduction": {"tests": 2, "errors": 2}, "host_disposition": "reject", "terminal": {"state": "failed", "version": 31}}, - "final_runtime_repair": {"run_id": "33c64397-3a57-42e7-82d5-773d50f466ee", "status": "running", "target": "6375955f66ce87a541fdfbf60db6eb0fcd30a55e", "scope": "NUL-safe filename discovery plus three unavailable-interpreter wheel-test helpers and regressions", "required_full_gate": "Explicit existing offline DEVSQUAD_BUILD_PYTHON, isolated HOME and strict ResourceWarning; no installation-test skip substitution"}, + "final_runtime_repair": {"run_id": "33c64397-3a57-42e7-82d5-773d50f466ee", "status": "awaiting_host", "version": 23, "target": "6375955f66ce87a541fdfbf60db6eb0fcd30a55e", "candidate": "af994e948eb38764884d27fd61665efa776cb47b", "candidate_sha256": "efb502a38972a97aba89cce371b398e2ac7e7dd1b6dd9b0a8fa83a51d98d9f5c", "writer": "verified native claude-sonnet-5", "reviewer": "verified native gpt-6.1-sol/low, read_only", "review": "clean", "checks": "all_three_passed_with_verified_unchanged_integrity", "full_core": {"tests": 475, "unittest_seconds": 460.153, "monotonic_seconds": 460.51592204100007, "utc_seconds": 460.515864, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "result": "pass"}, "accept_allowed": true, "checks_artifact_sha256": "5b543be7cb05ed8dbf7e66c60aee12fba16e8401332da5815f14569112fe156e", "operator_cases": "README and symlink negatives plus Unicode positive pass", "scope": "NUL-safe filename discovery plus three unavailable-interpreter wheel-test helpers and regressions", "required_full_gate": "Explicit existing offline DEVSQUAD_BUILD_PYTHON, isolated HOME and strict ResourceWarning; installed-wheel gates execute"}, + "final_combined_review": {"run_id": "ba14b743-9473-46fe-9e1f-d5b79fb032d4", "status": "running", "base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "af994e948eb38764884d27fd61665efa776cb47b", "scope": "five_changed_files", "checks": "mandatory diff, Bash and affected task-entry/wheel tests; no duplicate full gate"}, "delivery_retry": {"run_id": "8c116b7d-3416-4f80-a990-d610e2a5a558", "state": "failed", "version": 13, "worker_invocations": 1, "reason": "native_result_invalid", "output_bytes": 6183, "output_sha256": "0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485", "candidate_produced": false, "review_and_checks": "not_started", "cause": "unresolved; tool-free frozen-context replays pass including inherited regular-file stdin at EOF"}, "detached_auth_diagnosis": {"result": "reproduced_and_source_repaired", "original_environment": "PATH and frozen PYTHONPATH, without HOME/USER", "native_failure": "Synthetic assistant model, is_error true and not-logged-in terminal; no model usage", "home_only": "still_failed", "home_plus_user": "pass_one_Read_with_verified_claude_sonnet_5", "failure_stream_sha256": "f7f7a53de76012c54c4a0757aa346124553366bf6b1f0c120cb7288ce1491ab7", "home_user_stream_sha256": "1061f1ca7688fa178b4add47934c5341287d02c66a4ba468d91343398fa830d3", "red_regressions": {"tests": 2, "errors": 2}, "repair": "Allowlist HOME and USER in detached launch; classify synthetic native error terminals without claiming writer identity. API keys and provider overrides remain excluded.", "focused_gate": {"tests": 50, "seconds": 69.606, "result": "pass"}, "auth_priority_rerun": {"tests": 22, "seconds": 3.652, "result": "pass"}, "verification": "source_targeted_pass; managed_delivery_still_pending"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, From b6b924ff7e0976e06c395a16d72978d06d8adb33 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 18:08:00 -0700 Subject: [PATCH 164/197] WIP checkpoint: Record combined-review byte-filename finding and fenced managed revision (2026-10-01 18:08) --- docs/plans/engineering-team/RESUME.md | 19 +++++++++++++++++++ .../R8-installed-workflows-2026-10-01.json | 4 +++- 2 files changed, 22 insertions(+), 1 deletion(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 4eb6d82..ad49370 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -8,6 +8,25 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ### Passing managed gate — final combined review / integration remains +**Latest action: fenced revision in `33c64397-3a57-42e7-82d5-773d50f466ee`, +queued/version 26 at 01:05 UTC.** Full combined review `ba14b743…` completed +all required checks (25 affected tests in 24.390s plus Bash/diff, verified +integrity) but found one medium issue: `_git_entries_z` uses strict text +decoding and rejects valid non-UTF-8 Git filenames. It was host-rejected, +failed/version 22, not counted clean. Its exact base was 1781b89 and target +af994. The existing managed task allows the two affected files, so its scope +and budget were preserved while a hashed host revise requested bytes plus +UTF-8 surrogateescape and a real-Git non-UTF-8 documentation/valid-test case. +Do not apply the prepared af994 patch or accept the old snapshot with that +finding open. Wait for the revised exact candidate's review/full checks, +then repeat only the narrow combined-baseline review/affected checks. + +The managed run's original 1,800s wall budget began 00:46:23 UTC; it is not +extended by handoff/revision. If the new full gate cannot fit, retain its +truthful timeout/budget receipt and final candidate, and complete a fresh saved +exact-candidate branch-review/full-check run rather than editing old budgets, +receipts, packages or snapshots. No provider job is duplicated. + Current active full-combined branch review: **`ba14b743-9473-46fe-9e1f-d5b79fb032d4`**, exact base `1781b89` → target **`af994e948eb38764884d27fd61665efa776cb47b`**, read scope limited to the five changed files. Its diff/Bash/affected-wheel and diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index cd609e2..b97dc24 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "status": "managed_gates_pass_final_combined_review_and_install_pending", + "status": "passing_initial_managed_gate_byte_filename_revision_in_progress", "source_checkpoint": "c7ccc02", "installation": { "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", @@ -25,6 +25,8 @@ "corrective_gate": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "review": "verified native gpt-6.1-sol/low, clean; summary explicitly says full G4 diff inspected, frozen comparison 5cb9eab to 6375955", "bash_diff": "passed_with_verified_integrity", "full_core": {"tests": 471, "seconds": 473.7623341669996, "utc_seconds": 473.758578, "errors": 3, "failures": 0, "skips": 2, "unraisable": [], "result": "failed"}, "cause": "Three installed-wheel build_python helpers raise FileNotFoundError for nonexistent HOME-relative cached interpreter under deliberately isolated HOME instead of probing next candidate", "focused_reproduction": {"tests": 2, "errors": 2}, "host_disposition": "reject", "terminal": {"state": "failed", "version": 31}}, "final_runtime_repair": {"run_id": "33c64397-3a57-42e7-82d5-773d50f466ee", "status": "awaiting_host", "version": 23, "target": "6375955f66ce87a541fdfbf60db6eb0fcd30a55e", "candidate": "af994e948eb38764884d27fd61665efa776cb47b", "candidate_sha256": "efb502a38972a97aba89cce371b398e2ac7e7dd1b6dd9b0a8fa83a51d98d9f5c", "writer": "verified native claude-sonnet-5", "reviewer": "verified native gpt-6.1-sol/low, read_only", "review": "clean", "checks": "all_three_passed_with_verified_unchanged_integrity", "full_core": {"tests": 475, "unittest_seconds": 460.153, "monotonic_seconds": 460.51592204100007, "utc_seconds": 460.515864, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "result": "pass"}, "accept_allowed": true, "checks_artifact_sha256": "5b543be7cb05ed8dbf7e66c60aee12fba16e8401332da5815f14569112fe156e", "operator_cases": "README and symlink negatives plus Unicode positive pass", "scope": "NUL-safe filename discovery plus three unavailable-interpreter wheel-test helpers and regressions", "required_full_gate": "Explicit existing offline DEVSQUAD_BUILD_PYTHON, isolated HOME and strict ResourceWarning; installed-wheel gates execute"}, "final_combined_review": {"run_id": "ba14b743-9473-46fe-9e1f-d5b79fb032d4", "status": "running", "base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "af994e948eb38764884d27fd61665efa776cb47b", "scope": "five_changed_files", "checks": "mandatory diff, Bash and affected task-entry/wheel tests; no duplicate full gate"}, + "combined_review_finding": {"run_id": "ba14b743-9473-46fe-9e1f-d5b79fb032d4", "verdict": "findings", "finding": {"id": "g4-filename-decoding", "severity": "medium", "cause": "Strict text=True decoding rejects valid Git non-UTF-8 filenames before test filtering"}, "required_checks": "all_passed_with_verified_integrity", "affected_tests": {"tests": 25, "seconds": 24.390, "result": "pass"}, "host": "reject", "terminal": {"state": "failed", "version": 22}}, + "byte_filename_revision": {"run_id": "33c64397-3a57-42e7-82d5-773d50f466ee", "host": "revise", "state_at_submission": "queued", "version": 26, "repair_requested": "Read NUL Git output as bytes with UTF-8 surrogateescape; add real non-UTF-8 documentation plus valid-test regression", "scope_and_budget": "unchanged_frozen_contract", "remaining_gates": "new candidate native review/full suite plus final combined scope review, integration and final install"}, "delivery_retry": {"run_id": "8c116b7d-3416-4f80-a990-d610e2a5a558", "state": "failed", "version": 13, "worker_invocations": 1, "reason": "native_result_invalid", "output_bytes": 6183, "output_sha256": "0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485", "candidate_produced": false, "review_and_checks": "not_started", "cause": "unresolved; tool-free frozen-context replays pass including inherited regular-file stdin at EOF"}, "detached_auth_diagnosis": {"result": "reproduced_and_source_repaired", "original_environment": "PATH and frozen PYTHONPATH, without HOME/USER", "native_failure": "Synthetic assistant model, is_error true and not-logged-in terminal; no model usage", "home_only": "still_failed", "home_plus_user": "pass_one_Read_with_verified_claude_sonnet_5", "failure_stream_sha256": "f7f7a53de76012c54c4a0757aa346124553366bf6b1f0c120cb7288ce1491ab7", "home_user_stream_sha256": "1061f1ca7688fa178b4add47934c5341287d02c66a4ba468d91343398fa830d3", "red_regressions": {"tests": 2, "errors": 2}, "repair": "Allowlist HOME and USER in detached launch; classify synthetic native error terminals without claiming writer identity. API keys and provider overrides remain excluded.", "focused_gate": {"tests": 50, "seconds": 69.606, "result": "pass"}, "auth_priority_rerun": {"tests": 22, "seconds": 3.652, "result": "pass"}, "verification": "source_targeted_pass; managed_delivery_still_pending"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, From b81790706a3bdb76b5ac0c77519d3e5c0a93f18c Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 18:15:03 -0700 Subject: [PATCH 165/197] WIP checkpoint: Preserve failed no-op native revision and exact byte-filename recovery task (2026-10-01 18:15) --- docs/plans/engineering-team/RESUME.md | 25 +++++++++++++++++++ .../R8-installed-workflows-2026-10-01.json | 12 ++++----- 2 files changed, 31 insertions(+), 6 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index ad49370..5d68847 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -4,6 +4,31 @@ This file is the recovery entry point for a quota cutoff, interrupted task or ne ## Current position — October 1, 2026 +### Authoritative next action — repair the byte-filename finding + +At the latest checkpoint all owned jobs are terminal. Run `33c64397…` is +**failed/version 37**, `WORKFLOW_OUTPUT_INVALID`: its fenced revision returned +a verified native summary saying the original goal was already implemented, +made no edits, and candidate freezing correctly rejected the no-op. The +delivery worktree is clean at `af994e9`. The saved revision reason is present +and the profile-bound prompt hash passed validation; this is not evidence of +quota exhaustion or a dropped revision request. Revision stdout SHA-256: +`767af9da8716b5eda3922a9a299108fc99e67ad0d77c0efb015e714e79affffc`. + +The earlier first iteration genuinely passed Claude → independent Codex → +475 core tests, but it was not accepted because the exact combined review +found strict UTF-8 Git filename decoding. Retain that passing historical gate +and the later failed terminal receipt separately. Do not mutate either. + +Next: one fresh narrow issue-delivery task from `af994e9`, with the byte-safe +decoder and real-Git regression as its primary goal, only two writable files, +verified Claude writer/Codex reviewer and mandatory full/Bash gates. After a +clean accepted candidate, perform the exact five-file combined review from +1781b89, integrate via apply_patch, refresh the stable offline installation and +recheck Gemini CLI/MCP once. Broader R4–R7 and IDE UI acceptance remain open. + +### Historical continuation records (latest action above supersedes states below) + ### Active continuation — managed delivery failure remains unresolved ### Passing managed gate — final combined review / integration remains diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index b97dc24..b7a670c 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "status": "passing_initial_managed_gate_byte_filename_revision_in_progress", + "status": "initial_475_test_workflow_passed_revision_failed_no_changes", "source_checkpoint": "c7ccc02", "installation": { "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", @@ -21,13 +21,13 @@ "second_installation": {"source_checkpoint": "6f51db9", "release": "0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef", "source_digest": "7d7e408303b3f17043c0f413d2d9828feb390a73dacff0bce358921fe1802365", "payload_drift": false, "pip_check": "pass", "prior_releases_retained": true}, "auth_repaired_installation": {"source_checkpoint": "1781b89", "release": "0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef", "source_digest": "68d542f6e8ea69c9bdefdfd4625af45ea46a09fb7926964397cbad3d96c83f28", "payload_drift": false, "pip_check": "pass", "previous_releases_retained": true}, "auth_repaired_delivery": {"run_id": "45667697-1aa2-4a49-9737-2c4e637ebd26", "status": "cancelled", "version": 17, "candidate_produced": true, "candidate_commit": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_patch_sha256": "bb47b4b5522826609cabc335e0984b99ffd8311ad620a4e7cfe04c05a6906e26", "writer": {"model": "claude-sonnet-5", "verification": "verified", "model_source": "claude.stream.assistant.message.model", "reported_usage": {"input_tokens": 36, "output_tokens": 29404, "total_tokens": 29440}}, "review_and_checks": "no_completed_evidence", "reviewer_failure": "runner and child died without an exit receipt; exact cause unknown; zero capture/log bytes; dead ownership safely cancelled", "scope": "Genuine installed Claude implementation passed, but independent review/test acceptance did not complete"}, - "candidate_correction": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "status": "running", "requested_base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "candidate_commit": "6375955f66ce87a541fdfbf60db6eb0fcd30a55e", "candidate_sha256": "77d16f59fcf8594615b25f774b3f106f4883565ff83a613453469c045be91b8a", "operator_red_cases": ["README-only tests tree incorrectly selects Python discovery", "tests symlink incorrectly selects Python discovery"], "operator_recheck": "both pass on corrective candidate", "remaining_operator_finding": "Git-quoted Unicode test filename test_π.py is missed; reproduced offline, requires fenced revision", "actual_review_base": "5cb9eab81bf4f7ba01861ffb05193a8a67521200", "review_scope_note": "Frozen delivery review uses implementer baseline, not original requested task base; full combined branch review remains required", "mandatory_checks": "Bash plus spawn-safe full Python core runner, coordinator-owned"}, + "candidate_correction": {"run_id":"288ee6f9-f503-421a-b81a-d50ee0cf44d8","status":"failed","requested_base":"1781b89b1228dfca2ca9148df437af3ac04b971f","target":"5cb9eab81bf4f7ba01861ffb05193a8a67521200","candidate_commit":"6375955f66ce87a541fdfbf60db6eb0fcd30a55e","candidate_sha256":"77d16f59fcf8594615b25f774b3f106f4883565ff83a613453469c045be91b8a","operator_red_cases":["README-only tests tree incorrectly selects Python discovery","tests symlink incorrectly selects Python discovery"],"operator_recheck":"both pass on corrective candidate","remaining_operator_finding":"Git-quoted Unicode test filename test_π.py is missed; reproduced offline, requires fenced revision","actual_review_base":"5cb9eab81bf4f7ba01861ffb05193a8a67521200","review_scope_note":"Frozen delivery review uses implementer baseline, not original requested task base; full combined branch review remains required","mandatory_checks":"Bash plus spawn-safe full Python core runner, coordinator-owned","version":31}, "corrective_gate": {"run_id": "288ee6f9-f503-421a-b81a-d50ee0cf44d8", "review": "verified native gpt-6.1-sol/low, clean; summary explicitly says full G4 diff inspected, frozen comparison 5cb9eab to 6375955", "bash_diff": "passed_with_verified_integrity", "full_core": {"tests": 471, "seconds": 473.7623341669996, "utc_seconds": 473.758578, "errors": 3, "failures": 0, "skips": 2, "unraisable": [], "result": "failed"}, "cause": "Three installed-wheel build_python helpers raise FileNotFoundError for nonexistent HOME-relative cached interpreter under deliberately isolated HOME instead of probing next candidate", "focused_reproduction": {"tests": 2, "errors": 2}, "host_disposition": "reject", "terminal": {"state": "failed", "version": 31}}, - "final_runtime_repair": {"run_id": "33c64397-3a57-42e7-82d5-773d50f466ee", "status": "awaiting_host", "version": 23, "target": "6375955f66ce87a541fdfbf60db6eb0fcd30a55e", "candidate": "af994e948eb38764884d27fd61665efa776cb47b", "candidate_sha256": "efb502a38972a97aba89cce371b398e2ac7e7dd1b6dd9b0a8fa83a51d98d9f5c", "writer": "verified native claude-sonnet-5", "reviewer": "verified native gpt-6.1-sol/low, read_only", "review": "clean", "checks": "all_three_passed_with_verified_unchanged_integrity", "full_core": {"tests": 475, "unittest_seconds": 460.153, "monotonic_seconds": 460.51592204100007, "utc_seconds": 460.515864, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "result": "pass"}, "accept_allowed": true, "checks_artifact_sha256": "5b543be7cb05ed8dbf7e66c60aee12fba16e8401332da5815f14569112fe156e", "operator_cases": "README and symlink negatives plus Unicode positive pass", "scope": "NUL-safe filename discovery plus three unavailable-interpreter wheel-test helpers and regressions", "required_full_gate": "Explicit existing offline DEVSQUAD_BUILD_PYTHON, isolated HOME and strict ResourceWarning; installed-wheel gates execute"}, - "final_combined_review": {"run_id": "ba14b743-9473-46fe-9e1f-d5b79fb032d4", "status": "running", "base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "af994e948eb38764884d27fd61665efa776cb47b", "scope": "five_changed_files", "checks": "mandatory diff, Bash and affected task-entry/wheel tests; no duplicate full gate"}, + "final_runtime_repair": {"run_id":"33c64397-3a57-42e7-82d5-773d50f466ee","status":"failed","version":37,"target":"6375955f66ce87a541fdfbf60db6eb0fcd30a55e","candidate":"af994e948eb38764884d27fd61665efa776cb47b","candidate_sha256":"efb502a38972a97aba89cce371b398e2ac7e7dd1b6dd9b0a8fa83a51d98d9f5c","writer":"verified native claude-sonnet-5","reviewer":"verified native gpt-6.1-sol/low, read_only","review":"clean","checks":"all_three_passed_with_verified_unchanged_integrity","full_core":{"tests":475,"unittest_seconds":460.153,"monotonic_seconds":460.51592204100007,"utc_seconds":460.515864,"errors":0,"failures":0,"skips":2,"unraisable":[],"result":"pass"},"accept_allowed":null,"checks_artifact_sha256":"5b543be7cb05ed8dbf7e66c60aee12fba16e8401332da5815f14569112fe156e","operator_cases":"README and symlink negatives plus Unicode positive pass","scope":"NUL-safe filename discovery plus three unavailable-interpreter wheel-test helpers and regressions","required_full_gate":"Explicit existing offline DEVSQUAD_BUILD_PYTHON, isolated HOME and strict ResourceWarning; installed-wheel gates execute","gate_scope":"First iteration only; later fenced revision failed, no host acceptance"}, + "final_combined_review": {"run_id":"ba14b743-9473-46fe-9e1f-d5b79fb032d4","status":"failed","base":"1781b89b1228dfca2ca9148df437af3ac04b971f","target":"af994e948eb38764884d27fd61665efa776cb47b","scope":"five_changed_files","checks":"mandatory diff, Bash and affected task-entry/wheel tests; no duplicate full gate","version":22}, "combined_review_finding": {"run_id": "ba14b743-9473-46fe-9e1f-d5b79fb032d4", "verdict": "findings", "finding": {"id": "g4-filename-decoding", "severity": "medium", "cause": "Strict text=True decoding rejects valid Git non-UTF-8 filenames before test filtering"}, "required_checks": "all_passed_with_verified_integrity", "affected_tests": {"tests": 25, "seconds": 24.390, "result": "pass"}, "host": "reject", "terminal": {"state": "failed", "version": 22}}, - "byte_filename_revision": {"run_id": "33c64397-3a57-42e7-82d5-773d50f466ee", "host": "revise", "state_at_submission": "queued", "version": 26, "repair_requested": "Read NUL Git output as bytes with UTF-8 surrogateescape; add real non-UTF-8 documentation plus valid-test regression", "scope_and_budget": "unchanged_frozen_contract", "remaining_gates": "new candidate native review/full suite plus final combined scope review, integration and final install"}, - "delivery_retry": {"run_id": "8c116b7d-3416-4f80-a990-d610e2a5a558", "state": "failed", "version": 13, "worker_invocations": 1, "reason": "native_result_invalid", "output_bytes": 6183, "output_sha256": "0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485", "candidate_produced": false, "review_and_checks": "not_started", "cause": "unresolved; tool-free frozen-context replays pass including inherited regular-file stdin at EOF"}, + "byte_filename_revision": {"run_id":"33c64397-3a57-42e7-82d5-773d50f466ee","host":"revise","state_at_submission":"queued","version":26,"repair_requested":"Read NUL Git output as bytes with UTF-8 surrogateescape; add real non-UTF-8 documentation plus valid-test regression","scope_and_budget":"unchanged_frozen_contract","remaining_gates":"new candidate native review/full suite plus final combined scope review, integration and final install","terminal":{"state":"failed","version":37,"error":"WORKFLOW_OUTPUT_INVALID","message":"delivery candidate contains no changes"},"stdout_sha256":"767af9da8716b5eda3922a9a299108fc99e67ad0d77c0efb015e714e79affffc","diagnosis":"Saved revision reason present and normalized prompt hash validated; verified native writer summary incorrectly says original task already satisfied. Worktree unchanged at af994. Not a quota timeout."}, + "delivery_retry": {"run_id":"8c116b7d-3416-4f80-a990-d610e2a5a558","state":"failed","version":13,"worker_invocations":1,"reason":"native_result_invalid","output_bytes":6183,"output_sha256":"0886c28213a28db6617d2fa2e75a2cb0eb772007ba3006bdcfce598537fe9485","candidate_produced":false,"review_and_checks":"not_started","cause":"Reproduced missing HOME/USER detached saved-login boundary; fixed at 1781b89, subsequent genuine writer succeeds."}, "detached_auth_diagnosis": {"result": "reproduced_and_source_repaired", "original_environment": "PATH and frozen PYTHONPATH, without HOME/USER", "native_failure": "Synthetic assistant model, is_error true and not-logged-in terminal; no model usage", "home_only": "still_failed", "home_plus_user": "pass_one_Read_with_verified_claude_sonnet_5", "failure_stream_sha256": "f7f7a53de76012c54c4a0757aa346124553366bf6b1f0c120cb7288ce1491ab7", "home_user_stream_sha256": "1061f1ca7688fa178b4add47934c5341287d02c66a4ba468d91343398fa830d3", "red_regressions": {"tests": 2, "errors": 2}, "repair": "Allowlist HOME and USER in detached launch; classify synthetic native error terminals without claiming writer identity. API keys and provider overrides remain excluded.", "focused_gate": {"tests": 50, "seconds": 69.606, "result": "pass"}, "auth_priority_rerun": {"tests": 22, "seconds": 3.652, "result": "pass"}, "verification": "source_targeted_pass; managed_delivery_still_pending"}, "claude_compatibility_repair": {"status": "targeted_pass_full_and_independent_pending", "red": {"tests": 2, "failures": 1, "errors": 1}, "focused": {"tests": 62, "seconds": 70.381, "result": "pass"}, "repair": "Explicit argv separator in both launch paths; strict native stream parsing validates one session-correlated top-level writer model, retains auxiliary usage, and rejects contradictory, delegated, missing or uncorrelated identity. Legacy usage-only multiple-model results still reject.", "live_tool_free_smoke": {"result": "success", "writer_model": "claude-sonnet-5", "auxiliary_model": "claude-haiku-4-5-20251001", "stream_sha256": "db3231c624023e7dc3ed8e809e3e42c0687b266d7a8fb710d11c3f1e6dc94f13", "scope": "native format evidence, not managed delivery acceptance"}}, "claude_full_gate": {"revision": "3555a91", "tests": 458, "seconds": 467.018, "errors": 0, "failures": 0, "skips": 2, "unraisable": [], "result": "pass"}, From bb6999dfa103b395fc19cb5fa381842e1e36bfd3 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 18:16:57 -0700 Subject: [PATCH 166/197] WIP checkpoint: Record fresh bounded byte-filename task and real Git red regressions (2026-10-01 18:16) --- docs/plans/engineering-team/RESUME.md | 8 +++++++- .../evidence/R8-installed-workflows-2026-10-01.json | 1 + 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 5d68847..517abc8 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -20,7 +20,13 @@ The earlier first iteration genuinely passed Claude → independent Codex → found strict UTF-8 Git filename decoding. Retain that passing historical gate and the later failed terminal receipt separately. Do not mutate either. -Next: one fresh narrow issue-delivery task from `af994e9`, with the byte-safe +**Fresh task `360a4993-d016-406e-8bb4-ae6d57dfe6cf` is running/version 4**, +started at approximately 01:15 UTC with a new 1,800s budget, no revisions and +two worker invocations. The previous frozen budget/receipts are unchanged. +Private `observe-run.py` advances its ready candidate review, then stops at +handoff or a terminal state. Only one full core gate may run at a time. + +Next: finish this fresh narrow issue-delivery task from `af994e9`, with the byte-safe decoder and real-Git regression as its primary goal, only two writable files, verified Claude writer/Codex reviewer and mandatory full/Bash gates. After a clean accepted candidate, perform the exact five-file combined review from diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index b7a670c..7076012 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -1,6 +1,7 @@ { "schema_version": 1, "status": "initial_475_test_workflow_passed_revision_failed_no_changes", + "fresh_byte_correction": {"run_id": "360a4993-d016-406e-8bb4-ae6d57dfe6cf", "state_at_checkpoint": "running", "version": 4, "target": "af994e948eb38764884d27fd61665efa776cb47b", "scope": "two writable files, no revisions, two workers, 1800s wall budget; original frozen task unchanged", "mandatory_checks": "diff, Bash and full spawn-safe core suite with explicit offline build interpreter", "operator_red": {"case": "real Git index/tree non-UTF-8 documentation and test filenames", "result": "both raise UnicodeDecodeError on af994", "fixture_note": "Git plumbing inserts byte names into the index; macOS APFS rejects such working-tree names"}, "next": "finish candidate/native review/full gates, then narrow full-combined review, integration, offline install and one Gemini recheck"}, "source_checkpoint": "c7ccc02", "installation": { "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", From a52fc9cae1ba1d7a78bac4b064b5ba85b4b84f94 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 18:29:42 -0700 Subject: [PATCH 167/197] WIP checkpoint: Preserve accepted 477-test native workflow and concise final-review recovery note (2026-10-01 18:29) --- docs/plans/engineering-team/RESUME.md | 1029 +++-------------- .../R8-installed-workflows-2026-10-01.json | 5 +- 2 files changed, 134 insertions(+), 900 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 517abc8..9b7b972 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -1,903 +1,136 @@ # Resume DevSquad after an interruption -This file is the recovery entry point for a quota cutoff, interrupted task or new coding-agent session. Update it at each coherent checkpoint and before a long live probe. A pending milestone stays pending when its evidence is incomplete. - -## Current position — October 1, 2026 - -### Authoritative next action — repair the byte-filename finding - -At the latest checkpoint all owned jobs are terminal. Run `33c64397…` is -**failed/version 37**, `WORKFLOW_OUTPUT_INVALID`: its fenced revision returned -a verified native summary saying the original goal was already implemented, -made no edits, and candidate freezing correctly rejected the no-op. The -delivery worktree is clean at `af994e9`. The saved revision reason is present -and the profile-bound prompt hash passed validation; this is not evidence of -quota exhaustion or a dropped revision request. Revision stdout SHA-256: -`767af9da8716b5eda3922a9a299108fc99e67ad0d77c0efb015e714e79affffc`. - -The earlier first iteration genuinely passed Claude → independent Codex → -475 core tests, but it was not accepted because the exact combined review -found strict UTF-8 Git filename decoding. Retain that passing historical gate -and the later failed terminal receipt separately. Do not mutate either. - -**Fresh task `360a4993-d016-406e-8bb4-ae6d57dfe6cf` is running/version 4**, -started at approximately 01:15 UTC with a new 1,800s budget, no revisions and -two worker invocations. The previous frozen budget/receipts are unchanged. -Private `observe-run.py` advances its ready candidate review, then stops at -handoff or a terminal state. Only one full core gate may run at a time. - -Next: finish this fresh narrow issue-delivery task from `af994e9`, with the byte-safe -decoder and real-Git regression as its primary goal, only two writable files, -verified Claude writer/Codex reviewer and mandatory full/Bash gates. After a -clean accepted candidate, perform the exact five-file combined review from -1781b89, integrate via apply_patch, refresh the stable offline installation and -recheck Gemini CLI/MCP once. Broader R4–R7 and IDE UI acceptance remain open. - -### Historical continuation records (latest action above supersedes states below) - -### Active continuation — managed delivery failure remains unresolved - -### Passing managed gate — final combined review / integration remains - -**Latest action: fenced revision in `33c64397-3a57-42e7-82d5-773d50f466ee`, -queued/version 26 at 01:05 UTC.** Full combined review `ba14b743…` completed -all required checks (25 affected tests in 24.390s plus Bash/diff, verified -integrity) but found one medium issue: `_git_entries_z` uses strict text -decoding and rejects valid non-UTF-8 Git filenames. It was host-rejected, -failed/version 22, not counted clean. Its exact base was 1781b89 and target -af994. The existing managed task allows the two affected files, so its scope -and budget were preserved while a hashed host revise requested bytes plus -UTF-8 surrogateescape and a real-Git non-UTF-8 documentation/valid-test case. -Do not apply the prepared af994 patch or accept the old snapshot with that -finding open. Wait for the revised exact candidate's review/full checks, -then repeat only the narrow combined-baseline review/affected checks. - -The managed run's original 1,800s wall budget began 00:46:23 UTC; it is not -extended by handoff/revision. If the new full gate cannot fit, retain its -truthful timeout/budget receipt and final candidate, and complete a fresh saved -exact-candidate branch-review/full-check run rather than editing old budgets, -receipts, packages or snapshots. No provider job is duplicated. - -Current active full-combined branch review: **`ba14b743-9473-46fe-9e1f-d5b79fb032d4`**, -exact base `1781b89` → target **`af994e948eb38764884d27fd61665efa776cb47b`**, -read scope limited to the five changed files. Its diff/Bash/affected-wheel and -task-entry checks are mandatory; no duplicate full gate is declared. - -Managed delivery **`33c64397-3a57-42e7-82d5-773d50f466ee`** is now awaiting -host/version 23 with **accept_allowed true**. Actual Claude Sonnet 5 writer -and independent native gpt-6.1-sol/low reviewer are verified, review clean. -All three checks pass with unchanged/verified candidate integrity: diff, -227 Bash assertions, **475 core tests in 460.153s**, two optional-SDK skips, -zero errors/failures/unraisable diagnostics (UTC/monotonic ~460.516s agree). -The explicit build interpreter caused installed-wheel gates to run. All -operator README/symlink/Unicode cases pass. This is a genuine model workflow, -not fixture substitution. It is not terminal until fenced host acceptance. - -Next: inspect combined review's exact baseline/target and all required checks; -accept both saved runs only if clean/passing, then apply the prepared full G4 -diff from 1781b89 to af994 (five files) to the main project checkout, verify -blob hashes against the accepted candidate, affected/Bash gates and checkpoint. -Refresh the stable installation safely, verify drift/idempotence/doctor/SDK, -then one Gemini CLI/MCP status recheck against the final installed launcher. -Earlier failures below remain history; broader R4–R7 remain open. - -**Current active run: `33c64397-3a57-42e7-82d5-773d50f466ee`**, launched via -the installed stable `squad fix`, original base `1781b89`, retained target -`6375955f66ce87a541fdfbf60db6eb0fcd30a55e`. Claude is repairing NUL-safe -Unicode/tab/newline test filenames and three `InstalledWheel*.build_python` -helpers in `test_cli.py`, `test_handoff_store.py`, `test_mcp.py`: unavailable -HOME-relative interpreter candidates must catch OSError and try the next -candidate. This new task freezes all five allowed files; the previous run's -immutable scope/checks were not widened. The full check explicitly sets -`DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3.12` -so wheel gates run, not silently skip. Keep check HOME isolation and identity -validation strict. No API fallback/settings/reset changes. - -The writer finished candidate **`af994e948eb38764884d27fd61665efa776cb47b`**, -SHA-256 `efb502a38972a97aba89cce371b398e2ac7e7dd1b6dd9b0a8fa83a51d98d9f5c`, -five changed paths exactly as authorized. Unicode/README/symlink operator -fixtures now all pass. Native reviewer/check worker is running version 13. -The final full-combined review can use a prelaunch public branch-review task -with read scope limited to those five files, exact base 1781b89 → final target, -mandatory Bash plus affected tests with explicit offline build interpreter, -and no duplicate full gate. Do not broaden a frozen task or claim finished -checks before the saved packet exists. - -The prior corrective run `288ee6f9…` completed a verified native Codex clean -review, explicitly reporting inspection of the complete G4 diff from 1781b89, -although its frozen comparison remains 5cb9eab → 6375955. Bash/diff checks -passed with verified candidate integrity. Full core gate: **471 tests, -473.762s, 3 errors, zero failures/unraisable, two optional-SDK skips**. -All three errors are the missing build-interpreter candidate under isolated -HOME; two were reproduced immediately by a focused offline check. No pass is -inferred. The operator used exact artifact-bound host **reject**, terminal -**failed/version 31**, retaining all receipts and the candidate. - -Next: finish current Claude → native Codex → mandatory tests; verify the -previous negative and filename positive fixtures; accept only a passing saved -candidate. Run one bounded saved full-combined branch review from 1781b89 to -the final candidate (no duplicate full gate), then integrate that full diff via -apply_patch, affected/Bash gates, checkpoint, final safe install and Gemini -CLI/MCP recheck. Current installation remains `68d542f6e8ea`, source HEAD is -clean except this recovery/evidence update. Private host helper is -`/Users/Dikshant/.devsquad/private-probes/r8-installed-workflows-20261001/complete-handoff.py`; -it revalidates artifact hashes and does not print claims. Do not reuse the old -native Claude handoff's prior claim for a new run. - -**Latest active run: `288ee6f9-f503-421a-b81a-d50ee0cf44d8`.** The prior run -`45667697…` produced a verified Claude Sonnet 5 implementation and frozen -candidate `5cb9eab81bf4f7ba01861ffb05193a8a67521200`, but its reviewer runner -and child died without an exit receipt. The exact cause is unknown; stdout, -stderr and supervisor log were empty. Both process identities are dead and -public orphan cancellation terminalized it **cancelled/version 17**. Preserve -that truthful receipt and candidate; no review/test pass is inferred. - -The operator's isolated real-Git assessment found that `_has_tracked_files` -wrongly selects Python discovery for README-only test directories and a -`tests` symlink. The new bounded run starts from the retained candidate, repairs -these cases with Claude, and asks Codex to inspect the **whole** G4 diff from -`1781b89` to the new candidate. The review focus explicitly forbids running -tests/nested supervisors; the coordinator owns the mandatory Bash/full Python -checks. Codex reports ordinary usage allowed, not a current quota block. -Installed SDK transport rerun passes **9 tests in 3.191s**. Do not duplicate a -full core gate or modify frozen runtime packages/worktrees while this runs. -Next: inspect this run's final review/check receipts, accept only if all gates -and the reproduced cases pass, integrate the full accepted G4 diff, checkpoint -and refresh the final installation; then recheck Gemini CLI/MCP once. - -The corrective Claude writer finished with candidate -`6375955f66ce87a541fdfbf60db6eb0fcd30a55e`; both README/symlink red cases now -pass. The reviewer/check worker is running version 13, with the coordinator's -Python regression runner observed active. **Actual frozen review base is -`5cb9eab`, not `1781b89`**: delivery preparation uses its implementation -baseline even though the requested task base/focus names the original base. -Do not claim whole-combined-diff review from that receipt alone; supplement it -with a real branch review `1781b89` → final candidate before integration. -An additional real-Git fixture reproduces missed Unicode `test_π.py` because -the current parser reads quoted newline `ls-tree` output. After this gate, -use a fenced host **revise**, request NUL-safe regular-blob filename discovery -plus Unicode/tab/newline positive regressions, and finish the revised candidate -through independent review and mandatory tests. Do not accept the known issue -or duplicate the running full gate. No frozen runtime file has been modified. - -The auth repair is committed at `1781b89` and installed as -`0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef`; payload drift is false, -`pip check` passes, and previous releases remain recoverable. A fresh genuine -G4 delivery was launched: **`45667697-1aa2-4a49-9737-2c4e637ebd26`**, initially -running/version 4. It freezes the repaired source, scoped Claude Sonnet writer, -independent Codex gpt-6.1-sol/low adversarial review, Bash check and mandatory -spawn-safe full Python core runner (900s check timeout). Do not start a second -full core gate or edit its frozen candidate. Inspect saved status/evidence, -finish the fenced host disposition if eligible, then integrate only accepted -source changes and refresh the final installation. Managed delivery is not -yet a pass. The older installation/failures below are retained history. - -HEAD `6f51db9` is installed as `0.1.0-py31214-7d7e408303b3-mcp-a26bc88afbef`, -with payload drift false and `pip check` passed. The second genuine G4 delivery, -`8c116b7d-3416-4f80-a990-d610e2a5a558`, failed at version 13 with one -implementer invocation and `native_result_invalid`; no candidate, reviewer or -mandatory checks were produced. Its 6,183 native output bytes have only a safe -digest/typed failure receipt, not raw logs. Preserve both failed runs. - -A private tool-free replay passes even with the snapshot file at EOF on fd 0; -input inheritance is **not the cause**. One-Read replays pass under the operator -environment, but reproduce failure under the actual minimal detached -environment: a synthetic assistant model and a native `is_error: true`, -not-logged-in terminal, zero model usage. HOME alone still fails; HOME plus USER -restores the saved subscription login and passes the same one-Read replay. -No other environment variables or API credentials are needed. Raw output stays -private; redacted event/identity metadata is recorded in installed evidence. - -The repair preserves only HOME and USER alongside PATH/frozen PYTHONPATH at -the detached boundary. Native error terminals are classified before requiring -writer identity; a synthetic not-logged-in stream remains unverified and -reports AUTH_ERROR instead of generic CLI_ERROR, ahead of rate banners. Two red -regressions reproduced the omissions before source repair. The 50-test affected -gate passes in 69.606s; 227 Bash assertions and generated reference/diff pass. -The final auth-priority refinement passes a 22-test focused rerun in 3.652s. -Next: checkpoint after affected/Bash gates, refresh safely, then retry genuine G4 -Claude implementation → independent Codex review → mandatory tests under a -new run. Do not claim managed delivery passed or repeat completed host proofs. - -### Latest continuation — Claude launch/stream repair, Grok and Gemini proof - -The frozen stream/framing gate at `3555a91` passes **458 tests in 467.018s**, -two optional-SDK skips, zero errors/failures/unraisable diagnostics. Independent -native Codex review is clean with all three declared checks passed. A real -Claude Code MCP lead claimed, renewed and completed that saved review run -`b9501b55-1c65-4b32-a03f-1087cca8fefc`: **succeeded, version 23**. Earlier -probe failures are retained: empty builtins left MCP pending; a later model -falsely alleged a hash mismatch. Exact machine comparison confirmed all refs -matched before the successful fenced completion. ToolSearch plus only the -three scoped MCP tools works, without file/shell/delegation access. - -The real native stream additionally labels provider `firstParty`. It now maps -to Anthropic only for verified Claude 2.1.220, preserving its raw label and -rejecting other/unverified provider labels. The **63-test focused gate passes -in 70.391s**, 227 Bash assertions/reference/diff pass, and the saved real stream -now decodes to verified Sonnet 5 with Haiku usage retained. This mapping follows -the 458-test gate; the mandatory full suite on the upcoming genuine G4 candidate -must include it, and that independent delivery review will inspect the mapping. -Next: checkpoint, refresh the actual installation to these tested launch -repairs, retry G4 in a new saved run, and finish candidate review/checks plus -host disposition. Do not repeat Grok/Gemini or the completed Claude handoff. -Those live MCP proofs used the first refreshed release; MCP service source is -unchanged in the subsequent Claude worker repairs. Broader R4–R7 remain open. - -The installed G4 delivery attempt at `ea113d4`, run -`8a8a4548-13c1-4a2f-9190-870112f63ef2`, failed before implementation, version -13, with zero native reported tokens. The real cause is the variadic Claude -`--tools` option consuming a trailing prompt. Both launch paths now add `--`; -two red regressions reproduced the problem before repair. The tool-free native -smoke succeeds with that framing, but current Claude reports auxiliary Haiku -usage alongside the Sonnet writer. Strict stream evidence now requires all -top-level assistant messages to identify one concrete model under the final -session, retains all auxiliary usage, and revalidates on import. Usage-only -multi-model JSON still fails closed. Contracts updated; **62 focused tests pass -in 70.381s**, 227 Bash assertions and generated reference/whitespace pass. -Next: checkpoint, bounded independent Claude framing/identity review plus one -frozen full gate; refresh to these repairs before retrying G4 under a new run. -The actual installed release is still `674889018f98` and predates this repair. - -Grok 0.2.111 was refused with HTTP 426. The user explicitly authorized its -normal local update; 1.0.46 stable is installed with the old executable backed -up privately. Native `grok-4.7-build` actually called DevSquad status through -MCP and observed the exact failed run/version 13. Two bounded successful status -smokes are recorded, not hidden; no API fallback/settings changes. Gemini -3.8 Flash Low via Antigravity 1.2.13 also actually called the same updated -installation's MCP status and observed that exact run/version. Its first -unitless timeout was rejected at argument parsing; the corrected 120s call -succeeded. IDE UI remains permission-denied, distinct from CLI/MCP proof. -These proofs are in [installed workflow evidence](evidence/R8-installed-workflows-2026-10-01.json). - -### Latest continuation — upgrade review passed, full-gate clock fixture repair - -The actual local installation is now safely refreshed to -`0.1.0-py31214-674889018f98-mcp-a26bc88afbef` using the existing Python 3.12.14 -and locked offline MCP 2.2.0 wheelhouse. Reinstall is unchanged, all payload -drift flags are false, `pip check` passes, all four registrations match without -changes, and **22 installed-SDK tests pass in 5.585s with no skips**. The -schema-13 ledger had no active runs; a private SQLite backup and the old release -are retained. First new-release access migrated to schema 15 and read an old -succeeded result unchanged. See [installed workflow evidence](evidence/R8-installed-workflows-2026-10-01.json). -Next: genuine bounded G4 repair through installed `squad fix`, Claude writer, -independent Codex review and mandatory core checks; then actual Claude MCP -handoff plus Grok/Gemini status probes against that same updated installation. -Do not repeat the unchanged full gate or upgrade review. No model probe is -currently live; record its run ID before a handoff/interruption. - -At `6874de7`, the bounded native upgrade follow-up is clean. Both required -checks passed with verified unchanged candidate integrity; the accepted run -`3c89bb19-64ec-4600-9080-436df89edcfb` terminalized succeeded, version 22. -The proper spawn-safe full gate stopped failfast after 228 tests in 313.702s: -one lifecycle eligibility error, zero failures and zero unraisable diagnostics. -The original reader cause is `outcome observed_at exceeds allowed clock skew`. -The test injected module-import time, which ages beyond five minutes in the -full suite; a ten-minute-old clock reproduces it while the isolated case passes. -Use current runtime clocks in those fixtures; production validation remains -unchanged. An explicit regression keeps the stale-clock rejection strict. - -The focused lifecycle/eligibility gate passes 20 tests in 128.937 seconds; -all 227 Bash assertions, generated reference and whitespace checks pass. -The frozen full gate at `dc68944` now passes **455 tests in 485.168 seconds**, -with two optional-SDK skips, zero errors/failures and zero unraisable diagnostics. -UTC 485.364056s and monotonic 485.360219s agree. Retain the failed 228-test -clock-fixture gate as history; no additional production validation was relaxed. -Next: checkpoint, safely refresh the actual installation using the existing -Python 3.12/offline MCP wheelhouse, run installed SDK tests and prove Claude -delivery/handoff, Grok and updated Gemini/Antigravity. Use a bounded genuine G4 -check-discovery repair for the Claude implementation → Codex review → tests -workflow. The private offline probe reproduces both wrong-checkout discovery -and missing Python core checks. No provider job is currently live. The actual -installation still points at the schema-13 release. - -Privacy warning: Antigravity's global native MCP listing unexpectedly printed -an unrelated StitchMCP credential. It is not repeated or saved in evidence; -the user was advised to rotate it. Capture/filter future listings to DevSquad -only. Do not alter unrelated credentials/settings or bypass the denied IDE UI. - -### Current continuation — R3 audit repairs and safe schema-update deferral - -The two independent findings now have source repairs and red regressions. -Delivery evidence is decoded against the immutable prelaunch task, hash-bound -implementation imports, candidate/patch artifacts and candidate-ready events. -Review imports/checks must equal captured evidence, and independent identity -is revalidated. Revision prompts are reconstructed from recorded handoff -decisions, not unchecked final mutable snapshots. Failed writer/reviewer output -requires the exact terminal failure receipt. Qualification cannot substitute -caller-provided latency/usage ratios for missing saved paired measurements; -finite measurement gates remain blocked when those measurements are unknown. -Trusted Python checks suppress bytecode and redirect caches to their isolated -temporary HOME, without broadening candidate integrity allowances. - -The initial 22-test targeted reader/eligibility/check gate passed in 71.034 -seconds. The next 34-test gate had one test-helper error (an unpaired failed -arm has no paired case verdict), and the subsequent 19-test gate had one -assertion mismatch: the public replay correctly reports stale evidence rather -than exposing the lower-level failed-receipt reason. Both assertions are now -corrected; neither failed run is represented as a pass. The real historical -schema-13 installation test passed: the update defers without switching the -launcher, the old active run cancels, the queued old run resumes/completes, -then the update migrates to schema 15 and preserves readable results/releases. - -The normal-entry promotion fixture now uses an actual fake-native Codex -protocol, frozen adapters and execution fingerprints instead of pretending -fixture output is native. Its assertions pass after repairing cleanup order. -This is an offline public-chain proof, not real-model qualification. - -**Next action:** finish the affected regression gate, checkpoint, then run one -frozen full core integration gate and a bounded independent repair re-review. -Do not repeat the broad audit. Inspect/reconcile the real old ledger before -installing. The actual local installation remains unchanged; Claude/Grok and -updated Gemini/Antigravity live proofs remain pending. R4 catalog/quota, R5 -public trials/outcomes, R6 terminal UX and R7 Council remain separately open. - -The final affected gate passes **37 tests in 37.167 seconds**, including the -historical installed upgrade and fake-native normal-alias promotion proof. -All **227 Bash assertions in 11 files**, generated reference and whitespace -checks pass. A read-only inspection of the real default ledger found schema -13 with **zero nonterminal runs**; no reconciliation/cancellation is needed. -The installation still points at the original release. Checkpoint this slice -before the unchanged-source full integration run. Portable details are in -[audit repair evidence](evidence/R3-audit-repairs-2026-10-01.json). - -### Upgrade re-review and current full-gate rerun - -The bounded real Codex repair review against `db55d3f` → `e8cdf07` completed -in private runtime `r3-bounded-repair-review-20261001`, run -`fcc030af-26ec-423c-b462-b6fc0f8b55d1`. It found one high-severity upgrade -activation/admission race, not a separate R3 evidence/ratio finding. The -normal Bash check passed with **verified candidate integrity and no changed -inputs**, confirming the bytecode repair. The host rejected this review; -terminal state is failed, version 22. Native usage was 390,975 input / 4,422 -output tokens (395,397 total), one worker invocation, unknown internal request -count. Re-review only the additional upgrade repairs, not the broad source. - -Activation now checks/swaps under the ledger lock **without migrating**. The -new selected release performs lazy transactional migration; database guards -reject late writes by already-open old clients. Injected activation failure -leaves the old selector/schema untouched and releases the lock. The historical -schema-13 active/queued continuation/cancel test and pre-opened old-client -admission test pass. The final 12-test upgrade/capacity gate passes in 9.109 -seconds. The initial helper extraction run had two errors because the historic -package lacks the new helper; the installer now uses its own helper with the -selected package's explicit supported schema version. The old schema-8 SQL -backfill is still tested separately, while public upgrade correctly defers. - -The full integration launch from stdin was **invalid and interrupted (exit -130)**: spawned subprocess tests cannot reopen ``. It is not a passing -gate or a proven runtime failure. A tracked spawn-safe runner now records -UTC/monotonic timing, forced collection and unraisable diagnostics. Use -`PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning -PYTHONPATH=plugin/core/src:test/core python3 scripts/run-core-tests.py`. -Run one unchanged-source full gate after this checkpoint, then bounded upgrade -re-review. The real installation is still unchanged and all requested live -installed gates are pending. Do not repeat the spent Jev pilot or use paid APIs. - -The subsequent evidence/process regression gate passes **32 tests in 137.768 -seconds**, including real spawned-process recovery tests from a valid module -entrypoint. All 227 Bash assertions and generated-reference/whitespace checks -pass again. No process from the invalid stdin gate remains. The next full -gate must use the tracked runner, not stdin. The Antigravity IDE computer-use -surface was denied by the tool's app permission gate; do not bypass it. Recheck -the existing supported CLI/MCP surface after installation, and distinguish any -unavailable IDE UI proof from a CLI receipt. - -### Latest runtime-repair slice — shared eligibility and explicit review - -R3b.1's unchanged-source gate at `a4a87fd` passed 427 tests (two optional-SDK -skips). The subsequent **partial R3b.2/R3c** slice now adds schema 15's -append-only evaluation revisions and one transaction-bound current-evidence -gate for evaluation replay, qualification/replay, new promotion, regression -rollback and qualification-backed catalog fallback. Exact completed binding -decision replay remains historical and does not mutate again. Qualifications -match the full tested candidate and frozen role/task-class context. Proposal -inputs carry current eligibility; legacy v1 evidence cannot produce authority. - -The two-test red baseline reproduced stale qualification (`ContractError not -raised`) and the missing explicit revision operation. Six new eligibility -tests passed in 32.393 seconds. Lifecycle positive fixtures now use real -offline workers and public host completion rather than SQL-terminalized empty -runs: 14 lifecycle/assignment tests passed in 22.413 seconds. Learning's -positive tests retain their original two-evaluation/one-held-out gate and now -use six distinct actual runs. The latest 23-test learning/eligibility/reader -gate passed in 91.779 seconds; 227 Bash assertions and generated reference -passed. No full integration gate or independent audit has run on schema 15. -The source checkpoints, failure history and limitations are recorded in -[R3b.2 partial evidence](evidence/R3b2-eligibility-partial-2026-10-01.json). - -The next compatibility slice now proves stale regression rejection followed -by an explicit valid rollback review, stale qualified rollback-target rejection, -catalog fallback skipping that target (or blocking without any eligible -predecessor), correction/qualification writer fencing, current-versus-stale -proposal output and schema-13 duplicated-outcome history remaining readable -but ineligible. Fresh reviewed qualifying evidence pins a new promotion to its -revision hash. CLI revision dispatch passes. Four public delivery arms each -executed implementation, independent fixture review and checks with unchanged -source HEAD/checkout. Failures in that new test helper (wrong fixture shape -and treating a candidate-ready status as final review readiness) were repaired; -they are not represented as runtime defects or passing gates. - -The final targeted revision/installed-wheel checks passed two tests in 8.102 -seconds; the packaged wheel contains/applies schema 15. All 227 Bash assertions -and generated-reference/whitespace checks passed. This does not prove an old -active daemon can survive a schema change. - -The full unchanged-source run at `5de6655` completed **441 tests in 372.506 -seconds, FAILED with one error and two optional-SDK skips**. The headless -delivery mutation test could not find `receipt.json`; its temporary runtime -was cleaned before diagnosis. That case passes unchanged (1 test, 2.760 -seconds), and all nine check-integrity tests pass together (19.194 seconds). -This does not erase the failed integration gate or establish its cause. The -test now reports saved status/artifact names if the missing receipt recurs. -The SQLite finalizer warning has a concrete test-owned candidate: the installer -survival test used a SQLite transaction context without closing the connection. -It now uses `contextlib.closing`; its targeted active-release test passes in -6.837 seconds with ResourceWarning promoted to error, forced collection and -zero unraisable exceptions. The complete suite must still confirm no warning. - -The unchanged-source rerun at `dc4e110` completed **441 tests in 374.998 -seconds, OK with two optional-SDK skips**, with forced collection and **zero -unraisable exceptions** (no SQLite warning). UTC/monotonic elapsed times agreed -at 375.115/375.116 seconds. Retain the earlier failed run; its missing-receipt -cause remains unproven, not represented as a source repair. All 227 Bash -assertions and generated reference passed before this checkpoint. - -**Exact next action:** finish bounded R3 independent audit and continue R4's -normal routing/catalog/quota connections. Native Codex discovery currently -fails closed: PATH is 0.135.0, the previously verified bundled executable moved -to `ChatGPT.app/Contents/Resources/codex-cli/bin/codex` and is now 0.159.2. -A non-generating 0.159.2 probe passed initialize, complete model/list, -account/read and rateLimits/read without settings changes or model requests. -Version/path compatibility and operation verification remain separate gates. -The new bundled layout/version now has an explicit manifest and registration -entry. Its regression first failed on the missing path; 61 M1/MCP/task-entry -tests then passed (two optional-SDK skips), and all 227 Bash assertions passed. -Before the next source slice, checkpoint and run one bounded, source-only -native Codex R3 audit against the frozen `672e383` → current range in a private -runtime. This is not an updated-installation or Claude-delivery proof. Retain -the exact reviewer/check artifacts and any findings; do not claim completion -without reviewing them. No paid fallback or global configuration change. - -### Independent R3 audit — findings open; R4 alias slice partial - -The real read-only Codex 0.159.2 / `gpt-6.1-sol` high review completed against -`672e383` → `9334190` in private runtime -`/Users/Dikshant/.devsquad/private-probes/r3-independent-audit-20261001`, run -`80fdd336-07e2-445c-8a08-5877cd6534df`. It reports **R3-001 (high)**: delivery -implementation/reviewer imports are optional and stdout is not semantically -validated; **R3-002 (medium)**: submitted latency/usage ratios are not bound to -saved measurements. Both require reproductions and repairs. The normal Bash -check exited 0 but was correctly invalidated after creating undeclared Python -bytecode. The host rejected the packet and the run is terminal `failed`, version -22. No clean audit or installed proof is claimed. Native usage was reported -as 1,642,079 input / 10,733 output tokens, one worker invocation; internal model -request count is unknown. Do not repeat this broad audit. Re-review only the -bounded repairs when ready. - -In-flight R4 work adds normal stable aliases, explicit pin provenance and -policy-matched incumbent lookup before discovery. Two red tests reproduced the -old concrete-role bypass/missing binding API. The 29-test task-entry/CLI gate -passes in 5.073 seconds. A new public promotion-to-normal-entry fixture fails -early because its synthetic declaration lacks native adapter evidence for its -Codex profiles; this is a test-helper issue, not a passing public chain. The -new fixture parameters and test are deliberately checkpointed as **partial**; -do not weaken native experiment provenance to make them pass. - -**Exact next action:** repair the two independent R3 findings, with truthful -delivery success/failure evidence and measured-or-unknown ratio regressions; -fix check bytecode generation without weakening candidate integrity. Then -finish R4's public normal-entry proof, scoped catalog/quota, R5/R6 and the safe -R8 installation/Claude/Grok/Gemini gates. Installed release remains unchanged. -The shared gate audit and bounded independent review remain required for R3 -closure. Preserve the original evaluations and completed decisions. Do -not install schema 15 before the old active/recoverable-run upgrade test; R4–R6 -and R8's Claude/Grok/Gemini installed proofs remain pending. No provider call, -installation refresh, purchase, global setting change or push occurred. - -### Review correction and next action - -Interrupted R3b.1 work newer than the planning checkpoint is now preserved as -an explicitly **partial** saved-run reader. `experiment_evidence.py` joins -immutable assignments, preparation fences, actual attempts, stream/artifact -hashes and outcomes; `Store.evaluate_learning_experiment` uses it for v2 and -rejects changed-evidence replay without rewriting the old receipt. -`Store.reserve_attempt` checks assigned trials against the frozen controlled -input before launch. The public fixture uses real offline workers and host -dispositions, with a test-only predeclared-assignment preparation seam; it is -not a production paired-trial controller or native model-quality proof. - -The imported-profile regression first failed (`ContractError not raised`) -and passed after semantic profile binding was added. Earlier focused tests -passed 59 cases in 43.570 seconds; the additional prelaunch-mutation test -passed separately. The current combined gate and checkpoint details are in -[R3b.1 partial evidence](evidence/R3b1-reader-partial-2026-10-01.json). -No full Python integration gate, installed refresh or provider call has run -for this reader slice. The independent follow-up review hit its usage limit; -do not claim an independent reader audit passed. - -The terminal-failure concern is now reproduced and repaired. A real offline -failed reviewer exited with empty stdout and captured stderr, without a -`failure` metadata key; the old reader rejected it as invalid JSON. The reader -now requires either imported successful review evidence (still strictly -decoded) or a hash-verified terminal receipt with the exact failed/cancelled -attempt, role and frozen profile. A forged failure label cannot hide missing -successful evidence. Receipt run/profile/status/list corruptions fail closed. -The 34-test focused regression set passed in 42.301 seconds; six additional -success-integrity and saved-evidence tests passed in 9.803 seconds. The Bash -gate passed all 227 assertions. Source fingerprints and red/green results are -saved in [failure-path evidence](evidence/R3b1-failure-path-2026-10-01.json). - -The unchanged-source core gate at `a4a87fd` completed **427 tests in 260.966 -seconds, OK with two optional-SDK skips**. UTC and monotonic wrapper elapsed -times agreed (261.147 seconds). The known SQLite finalizer warning appeared -and remains R5; the gate did not establish its repair. R3b.1 is source/offline -verified, with the independent reader audit still outstanding. - -**Exact next action:** implement R3b.2's shared current-evidence eligibility and -explicit append-only evaluation/review revisions; legacy/public compatibility -remains R3c and the production paired-trial controller remains R5. - -Gemini/Antigravity's September 29 live `squad_status` receipt proves read-only -MCP observation of the same terminal-created saved run. It does not prove -Gemini implementation/review worker execution or the updated installation. -No extra Gemini login action is currently recorded; keep the normal session -signed in. On October 1, a non-generating `claude auth status` check confirmed -`loggedIn=true` through normal Claude authentication. The user reports Grok -signed in as well; its installed non-generating diagnostics do not establish -authentication, so its bounded live operation remains unverified. No model -request was made for those checks. - -Local Jev `.env` setup is complete: the user filled `TYPESAFE_API_KEY` locally, -and the file retains private mode 600 and Git ignore protection; the tracked -`.env.example` is blank. The probe now accepts explicit `--env-file .env`, -never executes shell text, preserves exported-key precedence and reports -only key presence during dry runs. All 14 focused probe tests pass, including -one mocked request and no-network dry runs. The 227-assertion Bash suite, -generated reference, JSON and whitespace checks pass; `.env` is untracked and -ignored while the blank example is trackable. No full core rerun was needed -for this isolated probe/setup change; R3b.1's full core gate remains pending. -The October 1 live M6-D2 pilot has now completed exactly one request and no -retries after current official pricing/API revalidation. Jev reported the -pinned `jev-1.13.0`, 5,373 input and 1,762 output tokens, and 187 ms for this -single sample. Estimated input charge is $0.00022567, below the $0.01 ceiling; -the invoice was not inspected. Task-family and skill labels each matched 8/8; -execution-tier matched 6/8. T07/T08 were lower than their frozen expected tiers, -including a wrong T07 label with 0.92 confidence. Do not change gold labels or -treat confidence as proven correctness. - -M6-D2 closes only the bounded synthetic smoke. Runtime classification remains -off; no measured routing benefit or quality/adoption gate is claimed. The -cost/access Laya trigger did not fire, and this smoke had no numeric production -quality threshold. No Laya setup or further hosted request is authorized by -this result. **Do not repeat the pilot:** its one-request allowance is spent. -The private receipt is outside Git; its hash and portable summary are in -[Jev pilot evidence](evidence/M6-D2-jev-pilot-2026-10-01.json). R3b.1's exact -next action above is unchanged; later shadow/adoption comparisons require a -predeclared corpus, held-out gates and separately approved budget. - -The earlier requested action was a review and execution plan for SOL. The October 1 -planning pass inspected `b766f9e` / source `672e383` and updated -[SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), without runtime changes, -installation or provider calls. Start with **R3b.1** (saved-run reader), then -**R3b.2** (shared eligibility and explicit evaluation/review revisions), then -R3c's realistic public and historical-compatibility proof. The plan now maps -each lifecycle consumer to its acceptance gate, clarifies R5 trial entrypoints -and terminal projection coverage, and requires an active-run schema-upgrade -safety test (coexistence or explicit deferral) before R8 refreshes the -installation. These are next-work requirements, not newly completed gates. -The verified source baseline below is unchanged; do not rerun it merely -because the planning files changed. - -Planning verification: all eight R3a source/test SHA256 values still match the -saved evidence; backlog JSON, generated-reference and whitespace checks pass. -The initial Bash run was interrupted after sandbox-denied process inspection -caused three portable-timeout/cleanup assertions to fail; it is not a pass. -Its verified test tree was stopped (exit 137). The single rerun with required -process access passed all **227 assertions in 11 files**. No test remains live. -The full Python gate was not repeated for documentation-only changes. - -R3a is source/offline verified at `672e383`: global outcome-reuse rejection, -versioned profile/input/assignment contracts, separate corpus-versus-pair -fingerprints, schema-14 prelaunch assignment persistence under the preparation -fence, and strict normalized v2 chain evaluation. The original five reuse -regressions failed before the repair and now pass. The focused experiment, -learning and lifecycle gate currently passes 47 tests; nine targeted migration -checks, including installed-wheel upgrades, also passed. Bounded review found -omitted tested-role fallback policy and native execution identity; both now -have regressions and fixes, including predeclared per-arm execution hashes. -The independent two-test follow-up passed with no remaining concrete finding -in the fixes. The full unchanged-source gate ran **402 tests in 213.208 seconds, -OK with two optional-SDK skips**; UTC/monotonic wrapper timing agreed at -213.376 seconds. The known SQLite finalizer warning remains R5. All 227 Bash -assertions and the generated-reference check passed. See -[R3a evidence](evidence/R3a-provenance-contract-2026-10-01.json). No test is -still running; begin R3b without repeating this unchanged gate. - -**R3 remains in progress.** The partial reader above now connects v2 saved-run -evaluation, but its complete integration gate and failure-path audit remain -open. R3b must enforce one current-evidence gate for -replay/qualification/promotion/rollback/catalog fallback. Audit/read -compatibility for unsafe legacy v1 proposals and public realistic fixtures -remain R3c; the public paired-trial controller remains R5. Do not count current -legacy positive fixtures as proof of that integration. No live providers, -installation refresh or global settings changes occurred. - -R2 source/offline repair is verified at `ef98889`: strict native result parsing, -v2 import evidence, actual-model independence and durable failed diagnostics. -The 16 original worker regressions, 12 import/parser regressions, 19 existing -delivery tests and nine public native delivery/fallback/cancellation/legacy -tests have passed focused runs. Independent review identified two additional -gaps (actual durable-attempt profile binding and exact native-byte retention); -both were reproduced, repaired and independently rechecked with four passing -targeted regressions and no additional actionable finding. - -The first integrated gate before those last two fixes passed 363 tests with -two optional-SDK skips. The subsequent 367-test gate completed with **six -failures and two skips**, not a pass. Four failures show budget expiration or -15–17 minute UTC jumps during a 231-second monotonic suite. On October 1, all -six failed cases passed unchanged in 18.998 seconds; per-test UTC and monotonic -elapsed measurements agreed. This supports an environmental timing explanation, -but does not erase the failed gate. The final unchanged-source rerun completed -**367 tests in 215.886 seconds, OK with two optional-SDK skips**; its wrapper -measured 216.021 UTC seconds and 216.020 monotonic seconds. The SQLite cleanup -warning still appeared and remains assigned to R5. All 227 Bash assertions -and the generated-reference check passed. **R2 source/offline closure is -recorded; installed/live closure remains R8.** See -[R2 evidence](evidence/R2-observed-identity-2026-10-01.json). No gate is running. - -No live provider call, installed refresh or global provider-setting change -occurred. The original red baseline is retained in -[R2 baseline evidence](evidence/R2-identity-red-baseline-2026-09-29.json). -The planning handoff was saved at `b3de5c6`; the persistent implementation -goal subsequently resumed R3a as recorded above. Do not redo R2 or retry -blocked authentication. The reproduced two-outcome/three-case reuse is now -rejected, but R3's full saved-run provenance and eligibility repair remains. - -R1's source repair is verified after `399d93d`: the public regression first -reproduced four unsafe mutation paths (review/delivery × host/headless). The -repair adds check-boundary fingerprints, explicit permitted output paths, -non-overridable invalidation, durable mutation evidence and historical receipt -compatibility. The final core gate ran **330 tests successfully, with 2 optional -SDK skips**. The final independent bounded R1 audit found no remaining actionable -issues; all **227 Bash assertions** and the generated-reference check passed. -See [R1 evidence](evidence/R1-candidate-integrity-2026-09-29.json) for the -full verification record, source fingerprints and limitations. The installed -runtime is not yet refreshed; that remains R8 work. - -The review of `f4fa6577e2e891151231c9c7d3180be6e9e23faa` supersedes the earlier -claim that only credentials remain. M1/M2 remain accepted; M3's F1 source repair -is verified with installed refresh pending; M4 retains its real-Claude-host gate; M5–M7 have -independent repairs and integration work; C1 remains pending full-delivery -scope. R1 is the first implemented repair; do not confuse it with full closure. - -Execute [SOL-REVIEW-FOLLOWUP.md](SOL-REVIEW-FOLLOWUP.md), starting at **R3**: -bind independent experiment evidence to saved runs, frozen assignments, -profiles and paired inputs; protect replay and future qualification decisions -against stale or unverified evidence. The plan contains bounded source-review -feedback and acceptance tests. Continue with routing/catalog/quota (R4), learning -runtime connections (R5), and normal terminal/readiness/check discovery (R6). -R7 covers C1 and R8 covers installed/live closure. Authentication and the Jev -key block their specific live subgates, not the independent engineering work. - -The original September 29 review ran 317 Python tests successfully (2 optional-SDK -skips), 227 Bash assertions, generated-reference validation and an installed -payload comparison. An unclosed SQLite `ResourceWarning` still appeared and -is assigned to R5. Existing live receipts remain evidence of their exact runs, -not proof that the newly identified failure cases are safe. See the follow-up -plan for reproduction details and evidence limits, and -[backlog.json](backlog.json) for current work-package dependencies. - -### Preserved implementation checkpoints - -The following records describe earlier implementation and verification; -the review correction above governs current completion and next work. - -- Workspace: `/Users/Dikshant/Desktop/Projects/devsquad`. -- Build branch: `codex/engineering-team`. `main` remains the published runtime - baseline. Inspect current refs before acting; later build checkpoints may be - local and must not be discarded. -- M1 is accepted for bounded truthful invocation/preparation at `97a10f0`. - Its saved integrated Codex receipt SHA256 is - `324c2ce6154936ecf71a8bf913a195befd4251efc6dbb8bbab3dc0dbe1b86df8`; - the private receipt remains outside the repository under - `/Users/Dikshant/.devsquad/private-probes/native-codex-20260907T025419Z-97a10f0ae1e8`. -- M2 is accepted at `ddb6f51`. Its durable store, gated runner, cancellation, - recovery, schema-5 host handoffs and CLI/service operations pass 121 core - tests with `ResourceWarning` promoted to failure and 202 Bash assertions. - See [M2-STATUS.md](M2-STATUS.md) and the - [portable redacted evidence](evidence/M2-durable-runs-2026-09-16.json). -- Final independent M2 review found two P1 races and no other defect in its - bounded target. `ddb6f51` fixes both: lease authorization now samples time - after acquiring the SQLite write transaction, and public cancel resumes an - interrupted `recovery_cleanup`. Both have deterministic regressions. -- M3 was accepted at `1737667` and is now reopened for F1. The branch-review path includes frozen - routing/input, native and offline reviewers, trusted checks, fenced host and - headless lead disposition, complete waiting/terminal reports, cumulative - budgets, transactional pool capacity and bounded frozen fallbacks. The final - gate is 188 core tests with `ResourceWarning` promoted to failure plus 202 - Bash assertions. See [M3-STATUS.md](M3-STATUS.md) and the - [portable closeout evidence](evidence/M3-branch-review-2026-09-22.json). -- The bounded real public `branch-review` gate passed at `9478796` with - verified gpt-5.5/low, read-only ephemeral execution, one supported finding, - a passing required check, native-reported usage and all five terminal report - hashes. See the [portable redacted evidence](evidence/M3-native-codex-review-2026-09-17.json). - That live receipt remains the M3 provider gate; closeout used offline tests - and did not consume another provider turn. -- The independent M3 audit at `84deb77` found four runtime/reporting defects. - `1737667` fixes all four with direct regressions: pre-launch recovery no - longer consumes a fallback/budget slot, headless lead exhaustion terminalizes, - unknown capacity allows one unresolved trial, and later failure/cancellation - retains earlier attempts and dispositions. A requested follow-up agent rerun - hit the shared Plus limit; the 188-test complete gate is green after the fixes. -- M4 Plan 06-01 is complete at `643910d`. The optional official MCP Python - SDK is pinned and transitively locked at `mcp==2.2.0`; ordinary CLI and a - plain installed wheel remain dependency-free. `squad mcp serve` exposes the - eight saved-run operations with strict envelopes, bounded event pages and a - 16 KiB total artifact-preview cap. Worker-origin mutations are denied while - read-only inspection remains available, and all four legacy hooks honor the - worker/delegation guard. The gate is 200 core tests with `ResourceWarning` - promoted to failure, 208 Bash assertions, and 12 focused tests against the - installed official SDK. -- M4 Plan 06-03 has completed all independent work at `aa0fef5`. The stable - isolated runtime at `~/.devsquad/releases/0.1.0+aa0fef5` is registered in - all four real local host configs; `squad doctor` reports four installed and - four matching registrations, and a second setup pass made no changes. One - terminal-started saved run was inspected through the real Codex MCP host, - then claimed/completed through official stdio SDK clients with identical - ledger identity, a rejected competing claim and terminal receipt hashes. - Closing the MCP client while a detached worker ran did not terminate it. - The gate is 212 core tests, 208 Bash assertions and 22 pinned-SDK tests. - M4 remains **blocked**, not complete, because its exact gate still requires - a normally authenticated Claude Code host to perform the real handoff; the - labeled SDK client is deliberately not presented as that proof. See the - [portable redacted evidence](evidence/M4-local-mcp-2026-09-23.json). -- Last-observed provider readiness outside accepted M3: Claude CLI is not - logged in and Grok CLI authentication is expired. Antigravity 1.2.13 is - authenticated and its project-scoped `mcp(devsquad/squad_status)` grant has - now produced a live Gemini receipt. Do not retry the blocked Claude/Grok - paths, buy credits, use paid API fallback or change global provider settings. - Resume those gates only after the user completes the corresponding login. -- User wants the implementation orchestrated efficiently and preserved across - Plus-plan interruptions. Avoid recursive subagent fan-out: it consumed the - shared window rapidly without advancing M3. The recursively created M3 - planning agents all hit the same Plus limit; continue locally until shared - agent capacity is restored, then use only bounded leaf reviews. -- Full assignment remains **M1–M7 plus C1**, as specified in - [SOL-HANDOFF.md](SOL-HANDOFF.md). The current review work packages R1–R8 - replace the earlier credentials-only assessment of M5–M7. -- M5 Plan 07-01 has started at `d96e9e4`. The core and Bash compatibility - boundaries now include a Claude 2.1.220 headless adapter with structured - output, version-scoped model/effort preparation, explicit Read/Glob/Grep or - Edit/Write tool sets, strict empty MCP configuration, safe mode and no - blanket permission bypass. Its offline gate is 215 core tests, a fresh-wheel - content check and 220 Bash assertions. This is adapter conformance, not a - live Claude model receipt; normal Claude login remains required. -- M5 Plan 07-01 delivery workspace/candidate freezing is complete at `6a7e849`. - A detached run-owned implementation worktree now enforces declared write - scope and symlink containment, creates one coordinator-owned local commit, - returns stable commit/tree/patch/candidate hashes, replays idempotently and - rejects later mutation. Five focused tests prove the original checkout, - index, HEAD and remote refs remain unchanged. The complete gate is 220 core - tests and 220 Bash assertions. One first core run observed the pre-existing - coordinator-crash test return `live`; its isolated run and the complete rerun - passed unchanged. -- M5 Plan 07-01 is complete at `0e88d73`. The offline implementer now executes - inside the durable process/writer fence, validates frozen attempt evidence, - serializes candidate finalization across recovery importers and atomically - saves implementation, local commit and patch evidence before creating - separate review/check worktrees. Concurrent resume creates no second writer; - out-of-scope edits fail without a candidate. The complete gate is 224 core - tests and 220 Bash assertions. A reproduced macOS zombie-only process-group - ambiguity was fixed with a non-zombie inventory check and direct regression. -- M5 Plan 07-02 first-candidate review/check/handoff is verified offline at - `30b98df`. The durable implementer creates a frozen candidate, explicit - resume launches read-only review, separate trusted checks validate that - candidate, and the host receives an `issue-delivery` handoff bound to its - hash. The complete gate is 226 core tests discovered (suite OK, 2 optional - SDK skips) and 220 Bash assertions. This does not complete lead disposition, - revise-to-implementation, all terminal reports or the live two-harness gate. -- M5 Plan 07-02 bounded revisions are verified offline at `4c76887`, following - delivery accept/reject receipts at `a1199c6`. A saved `revise` transaction - now returns the writer fence to the delivery worktree, binds prior evidence - into the next prompt, creates a distinct child candidate and new review/check - worktrees, rejects stale live evidence and preserves both candidates plus all - successful worker attempts and dispositions. Required checks remain - non-overridable; revision/invocation exhaustion stops before a new writer; a - simulated prelaunch crash resumes exactly one repair writer; and headless - delivery acceptance uses its own fenced lead attempt. The gate is 237 core - tests (2 optional-SDK skips) and 220 Bash assertions. Delivery fallback, - failure/cancellation history, live Claude execution and the live two-harness - proof remain open. -- M5 was assessed as offline-complete at `f199cd2`; F1/F2/G4 now reopen that assessment. - `b7d90cc` freezes the real Claude implementation bridge, records observed - identity/session/usage, enforces the same-permission rate-limit fallback and - preserves delivery failure/cancellation history. `f199cd2` kills a live - revised-implementation supervisor and proves retained ownership, no duplicate - writer, successful reap and unchanged source checkout/remotes. The exact core - gate is 241 tests with 2 optional-SDK skips and ResourceWarning promoted to - error; the compatibility gate is 220 Bash assertions. Normal Claude login - remains required for the genuine Claude implementation → different-model - Codex review/check/disposition receipt, after the independent repairs. -- M6 shared-capacity work is verified through `d170c01`. Schema 9 persists - strict scoped observations and reservations, backfills active schema-8 - attempts and keeps ambiguous ownership in flight. Two separate projects - racing for an unresolved pool create one reservation; fresh exhausted weekly - evidence beats short-window availability; stale/estimated/incomplete data - remains unknown; reservation rechecks post-preflight changes; and routing - applies model/profile sublimits without widening eligibility. `squad capacity - observe --file FILE` and frozen/current status evidence are wired. The gate - is 255 core tests with 2 optional-SDK skips and 220 Bash assertions. See - [M6-STATUS.md](M6-STATUS.md). -- M6 outcome/report/experiment evaluation is verified through `edfb3f3`. - Schema 10 records append-only final and late outcomes with truthful attempt - contribution, and reports separate automatic, pinned and experimental - evidence with sample size and missingness. Schema 11 freezes one-variable - paired experiments, evaluates evaluation and held-out splits, retains every - failure and rollback target, rejects conflicting replay and never changes - active policy. The public path is `squad policy evaluate --experiment FILE`. - `98c6685` adds `squad learn propose --project PATH`, which writes - content-addressed local JSON/Markdown drafts from one consistent ledger - snapshot, verifies saved evidence hashes and returns explicit no-change when - evidence is absent. -- M6 profile lifecycle is verified through `ca4ee73`. Schema 12 stores - versioned templates, immutable concrete profiles, bounded qualification, - compare-and-swap bindings and immutable decision receipts. Promotions affect - new runs only; exact pins and frozen runs do not float. `4b27e0c` binds - ordinary regression rollback to a saved post-change held-out evaluation. - `398ae6a` scopes complete catalog drift to affected profiles while keeping - additions unqualified, and `ca4ee73` rolls a removed incumbent only to the - newest prior proven/qualified profile under the same template or blocks - without mutation. The exact gate is 275 core tests with 2 optional-SDK skips - and 220 Bash assertions. These component checkpoints do not close the - evidence-integrity and missing runtime connections now assigned to R3–R5. -- M6-D1 is verified at `87fa9cf`. The optional decision helper defaults to off; - shadow records without changing execution, and advisory can only reorder the - deterministic router's already-eligible profiles under reviewed gate - evidence. Schema 13 fences and caches calls, prevents duplicate paid calls on - replay, records launched-unknown outcomes as indeterminate, exposes run - accounting in status and stores only hashes/byte counts for task evidence. - Malformed, unknown-ID, NaN, pin, permission/quality, drift, cancellation and - crash/resume cases fail closed. The frozen synthetic baseline explicitly - keeps runtime adoption off. The gate is 291 core tests with 2 optional-SDK - skips and 220 Bash assertions. M6-D2 was then blocked on the key; the October 1 - one-request smoke above resolves that measurement only. Laya remains - conditional on its declared trigger and runtime guidance remains off. -- M7 packaging, normal task entry and the currently available live surfaces are - verified through `b1d52ad`. The immutable standalone installer works without - Claude, performs offline exact-lock MCP installation with `pip check`, emits - clean JSON, migrates only a recognized legacy launcher, retains old releases - and is idempotent. The active installed release is - `0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef`; setup reports all four host - registrations unchanged and doctor is ready. Bundled Codex - 0.155.0-alpha.9.2 passed native initialize plus a complete seven-model - catalog and is preferred over PATH Codex 0.135.0. A terminal-created run was - observed through real Gemini/Antigravity and Codex MCP calls, then cancelled - from terminal. `85aa378` adds installed `squad review` and `squad fix` - commands that resolve exact commits, discover the Codex catalog without a - generation, embed hash-frozen routing and need no hand-written JSON. The - offline fix gate creates one isolated writer, freezes the candidate, runs - independent review and checks, preserves the source checkout and reaches - host handoff; `--wait` advances the saved candidate-review phase once. - `1553768`, `09e435c` and `9d70888` align structured-output recovery with - Codex 0.155's completed agent messages and retain only redacted structural - diagnostics. `b1d52ad` gives every trusted check a fresh temporary HOME so - detached checks work without exposing the user's real home or sharing state. - The exact `9d70888..b1d52ad` candidate then received a verified clean - gpt-5.5/low read-only review; both frozen checks passed and the artifact-bound - host acceptance terminalized succeeded. The gate is 317 core tests with 2 - optional-SDK skips, 227 Bash assertions, 8 focused installer tests and 22 - installed-SDK tests. See - [M7-STATUS.md](M7-STATUS.md) and the - [installed-runtime evidence](evidence/M7-installed-runtime-2026-09-29.json) - plus [normal-entry evidence](evidence/M7-normal-entry-2026-09-29.json) and - [live Codex review evidence](evidence/M7-live-codex-review-2026-09-29.json). - R4/R6 now identify remaining independent work; normal Claude login, renewed - Grok authentication and the installed two-model delivery are additional - external gates. -- The user's Jev/Laya request is evaluated in - [DECISION-CLASSIFIERS.md](DECISION-CLASSIFIERS.md). This source-backed plan - amendment adds M6-D1–D3: default-off contracts/baseline, a one-request capped +This is the authoritative recovery entry point, not a chronological chat log. +Compare it with Git status/recent commits and retain newer work. Earlier notes +remain recoverable in Git; detailed receipts and failed gates stay in evidence. + +## Current checkpoint — October 1, 2026 + +Workspace: /Users/Dikshant/Desktop/Projects/devsquad. +Branch: `codex/engineering-team`; never restart this build from main. +Source runtime repairs through `1781b89` are installed. The accepted G4 +candidate below is **not yet integrated or installed**. + +### Latest verified result + +Genuine saved issue delivery **`360a4993-d016-406e-8bb4-ae6d57dfe6cf`** is +**succeeded/version 31**, with fenced host acceptance: + +- Exact candidate: `f8c4f8c83f170870eb37d564e54eee6188fc233c`. +- Candidate SHA-256: `3f0fa0e86edd8a4c7b04bc1a2447d70b136e596cafe3624826f1f46c2fcee568`. +- Verified native Claude Sonnet 5 writer (Claude 2.1.220); independent verified + Codex gpt-6.1-sol/low read-only reviewer, clean, no findings. +- All three mandatory checks passed with unchanged verified integrity: diff, + 227 Bash assertions, **477 core tests in 455.860s**, two optional-SDK skips, + zero errors/failures/unraisable diagnostics. UTC/monotonic ~456.22s agree. +- Explicit offline DEVSQUAD_BUILD_PYTHON made installed-wheel gates execute. +- Real Git README/symlink/Unicode/byte-documentation/byte-test cases all pass. + Byte-name fixtures use index plumbing because APFS rejects those names. + +The final **complete-diff branch review** is running: +**`a6ecd887-5122-4281-b988-4d344a662e20`**, exact base +`1781b89b1228dfca2ca9148df437af3ac04b971f` → candidate `f8c4f8c…`. +Its scope is only the five changed files, with mandatory diff/Bash/affected +tests; no duplicate full gate. Private observe-run.py stops at its handoff. + +### Exact next action + +1. Inspect that combined review's packet and verified artifact hashes. Accept + only a clean review with all mandatory checks passed and unchanged integrity. +2. Integrate the accepted full five-file diff through apply_patch and verify + each source/test blob against `f8c4f8c…`; run affected/Bash/reference gates. + A prepared patch is cached in functions store, but regenerate from Git if + unavailable. Never edit a frozen run/candidate/receipt. +3. Checkpoint, safely refresh the immutable local installation offline using + existing Python 3.12.14 and MCP 2.2.0 wheelhouse; verify idempotence, drift, + pip check, doctor, matching host registrations and installed SDK tests. +4. Run **one** bounded Gemini/Antigravity CLI/MCP status recheck against the + final installed launcher and accepted run. Do not repeat completed Grok or + Claude host proofs. +5. Record final results/limits in backlog and installed evidence. Whole-plan + R3/R4–R7/C1 closure remains separate; use SOL-REVIEW-FOLLOWUP.md next. + +Only one full core suite may run at a time; freeze source/tests while it runs. +Use the tracked spawn-safe scripts/run-core-tests.py, never a stdin main. +Run bash test/run.sh to completion before every commit. Keep a clean tree; +use git-safety checkpoints, never stash. No goal is currently active. + +## Current local installation and host proof + +Stable launcher: /Users/Dikshant/.local/bin/squad. +Selected release: `0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef`. +Python 3.12.14/MCP 2.2.0, schema 15; previous releases and private pre-upgrade +SQLite backup retained. Source/plugin/installed payload drift is false. +Upgrade defers for old active/recoverable runs, swaps without migration under +the lock, then lazily migrates with old-client write guards. + +- Real Claude MCP lead claimed/renewed/completed + `b9501b55-1c65-4b32-a03f-1087cca8fefc`, succeeded/version 23. + This is CLI/portable handoff, not desktop local Code-tab UI proof. +- Grok user-approved normal updater installed 1.0.46 stable; old executable + backed up privately, login/settings preserved. Native grok-4.7-build actually + called DevSquad status. This is not automatic Grok writer/reviewer proof. +- Antigravity 1.2.13/Gemini 3.8 Flash Low actually called the same MCP status. + Existing project-only grant, plan/sandbox; final-install recheck remains. + IDE UI permission was denied; do not bypass it. +- Initial installed SDK gate: 22 passed/no skips; transport follow-up: 9 passed. + Final refreshed-install SDK gate still pending. + +## Repairs and failure history to preserve + +Full portable redacted record: +[evidence/R8-installed-workflows-2026-10-01.json](evidence/R8-installed-workflows-2026-10-01.json). + +- Claude argv variadic tools consumed the prompt: explicit -- separator fixed. +- Strict bounded native JSON/stream identity verifies one session-correlated + writer model, retaining auxiliary usage. firstParty maps to Anthropic only + for verified exact Claude 2.1.220; unknown serving revisions stay unknown. +- Detached login needs HOME plus USER. Minimal environment and HOME-only + reproduced synthetic not-logged-in output. Allowlist repaired; no API keys + or provider overrides inherited; native auth errors fail without identity. +- `45667697…` writer succeeded, but reviewer runner/child died without a + receipt; cause unknown. Dead ownership safely cancelled/version 17. +- `288ee6f9…` clean native review, but 471-test gate had three missing + build-interpreter helper errors. Host rejected, failed/version 31. +- `33c64397…` first iteration passed real 475-test workflow, but exact + combined review `ba14b743…` found strict UTF-8 byte-name decoding; rejected. + Its fenced Claude revision ignored the new issue and made no edits. Prompt + hash/reason were validated; correctly failed/version 37 for no changes. + It is **not** a quota timeout or accepted candidate. Fresh task above fixed it. +- G4 now uses exact target Git regular blobs/trees, actual tracked test*.py, + Bash plus core checks, required checks, exact-argv dedup and byte/NUL-safe + filename parsing. Three wheel-test helpers catch unavailable interpreter + candidates without relaxing isolation/dependency checks. + +## Broader plan / constraints + +M1/M2 remain accepted. R1/R2 source/offline repairs are verified; R3 saved-run +reader, eligibility/revisions, provenance/ratio and upgrade repairs have full +and bounded evidence, but final package-level closure audit remains open. +Normal aliases/public promotion proof exist; this does not close R4 catalog/ +quota, R5 public trial controller/outcomes, remaining R6 UX or R7/C1 Council. +Read backlog.json and SOL-REVIEW-FOLLOWUP.md for dependency/acceptance details; +do not restart the architecture exercise or weaken gates to mark these done. + +Jev .env is private/ignored and complete. Exactly one authorized pilot request +was spent; 8/8 family and skill labels, 6/8 tier labels including a confident +wrong tier. Runtime classification remains OFF; no adoption benefit or Laya +setup is proven. Do not repeat the pilot or silently use hosted/API fallback. + +No credit purchases/resets, paid API fallback, global AI settings, push/merge, +deploy or external messages are authorized. Raw provider output and credentials +stay outside Git. Native reported cost/usage is not an inspected subscription +invoice or a reliable number of Plus five-hour windows. + +Privacy: an earlier global Antigravity MCP listing exposed an unrelated +StitchMCP credential; user was advised to rotate it. Never repeat/store it. +Filter future diagnostics to DevSquad only; do not change unrelated credentials. + +Private bounded helpers are under +/Users/Dikshant/.devsquad/private-probes/r8-installed-workflows-20261001: +observe-run.py, inspect-handoff.py, complete-handoff.py, start-combined-review.py, +g4-candidate-cases.py and probe.py. Claims/raw logs remain private; never reuse +the completed b950 handoff's prior claim for a different run. Jev pilot, and local Laya fallback plus measured adoption. It prioritizes routing hints, skill/tool shortlists and context ranking, followed by failure triage, review attention and outcome labels. No weights/inference/API spending or diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index 7076012..05dd44c 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -1,7 +1,8 @@ { "schema_version": 1, - "status": "initial_475_test_workflow_passed_revision_failed_no_changes", - "fresh_byte_correction": {"run_id": "360a4993-d016-406e-8bb4-ae6d57dfe6cf", "state_at_checkpoint": "running", "version": 4, "target": "af994e948eb38764884d27fd61665efa776cb47b", "scope": "two writable files, no revisions, two workers, 1800s wall budget; original frozen task unchanged", "mandatory_checks": "diff, Bash and full spawn-safe core suite with explicit offline build interpreter", "operator_red": {"case": "real Git index/tree non-UTF-8 documentation and test filenames", "result": "both raise UnicodeDecodeError on af994", "fixture_note": "Git plumbing inserts byte names into the index; macOS APFS rejects such working-tree names"}, "next": "finish candidate/native review/full gates, then narrow full-combined review, integration, offline install and one Gemini recheck"}, + "status": "accepted_477_test_native_workflow_final_combined_review_running", + "fresh_byte_correction": {"run_id":"360a4993-d016-406e-8bb4-ae6d57dfe6cf","state_at_checkpoint":"running","version":4,"target":"af994e948eb38764884d27fd61665efa776cb47b","scope":"two writable files, no revisions, two workers, 1800s wall budget; original frozen task unchanged","mandatory_checks":"diff, Bash and full spawn-safe core suite with explicit offline build interpreter","operator_red":{"case":"real Git index/tree non-UTF-8 documentation and test filenames","result":"both raise UnicodeDecodeError on af994","fixture_note":"Git plumbing inserts byte names into the index; macOS APFS rejects such working-tree names"},"next":"finish candidate/native review/full gates, then narrow full-combined review, integration, offline install and one Gemini recheck","terminal":{"state":"succeeded","version":31,"host":"accept"},"candidate":{"commit":"f8c4f8c83f170870eb37d564e54eee6188fc233c","sha256":"3f0fa0e86edd8a4c7b04bc1a2447d70b136e596cafe3624826f1f46c2fcee568"},"writer":{"model":"claude-sonnet-5","verification":"verified","harness":"2.1.220 (Claude Code)","input_tokens":12,"output_tokens":12079,"total_tokens":12091},"review":{"model":"gpt-6.1-sol","effort":"low","verification":"verified","permission_policy":"read_only","verdict":"clean","input_tokens":90119,"output_tokens":481,"total_tokens":90600},"gates":{"checks":"all_three_passed_unchanged_verified_integrity","bash_assertions":227,"full_core":{"tests":477,"unittest_seconds":455.86,"monotonic_seconds":456.2232173750008,"utc_seconds":456.21989,"errors":0,"failures":0,"skips":2,"unraisable":[],"result":"pass"},"checks_artifact_sha256":"ee7d7c903872364b6c3172d2b68084f8e352578ef666564e58dcbd1a63985820"},"operator_recheck":"All five README/symlink/Unicode/byte-documentation/byte-test cases pass"}, + "byte_fix_combined_review": {"run_id": "a6ecd887-5122-4281-b988-4d344a662e20", "state_at_checkpoint": "running", "version": 4, "base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "f8c4f8c83f170870eb37d564e54eee6188fc233c", "scope": "five changed files", "checks": "mandatory diff, Bash and 27 affected tests; no duplicate full gate"}, "source_checkpoint": "c7ccc02", "installation": { "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", From 6d2e0ba39a51a263a7b93df5b40d2e8890cfe01c Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 18:32:24 -0700 Subject: [PATCH 168/197] WIP checkpoint: Integrate independently accepted exact-target and byte-safe discovery repair (2026-10-01 18:32) --- docs/plans/engineering-team/RESUME.md | 130 +----- .../R8-installed-workflows-2026-10-01.json | 5 +- plugin/core/src/devsquad/task_entry.py | 105 ++++- test/core/test_cli.py | 29 +- test/core/test_handoff_store.py | 39 +- test/core/test_mcp.py | 31 +- test/core/test_task_entry.py | 382 ++++++++++++++++++ 7 files changed, 569 insertions(+), 152 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 9b7b972..f9d11af 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -9,7 +9,9 @@ remain recoverable in Git; detailed receipts and failed gates stay in evidence. Workspace: /Users/Dikshant/Desktop/Projects/devsquad. Branch: `codex/engineering-team`; never restart this build from main. Source runtime repairs through `1781b89` are installed. The accepted G4 -candidate below is **not yet integrated or installed**. +candidate below is **integrated, not yet installed**; all five blob hashes +exactly match the accepted candidate. Local affected gate: 27 tests in 18.940s, +plus 227 Bash assertions, generated-reference and diff checks passed. ### Latest verified result @@ -27,27 +29,25 @@ Genuine saved issue delivery **`360a4993-d016-406e-8bb4-ae6d57dfe6cf`** is - Real Git README/symlink/Unicode/byte-documentation/byte-test cases all pass. Byte-name fixtures use index plumbing because APFS rejects those names. -The final **complete-diff branch review** is running: +The final **complete-diff branch review** is **succeeded/version 22**, accepted: **`a6ecd887-5122-4281-b988-4d344a662e20`**, exact base `1781b89b1228dfca2ca9148df437af3ac04b971f` → candidate `f8c4f8c…`. Its scope is only the five changed files, with mandatory diff/Bash/affected -tests; no duplicate full gate. Private observe-run.py stops at its handoff. +tests; no duplicate full gate. All checks passed with unchanged verified +integrity; 27 affected tests in 24.173s. Verified native Codex review is clean. ### Exact next action -1. Inspect that combined review's packet and verified artifact hashes. Accept - only a clean review with all mandatory checks passed and unchanged integrity. -2. Integrate the accepted full five-file diff through apply_patch and verify - each source/test blob against `f8c4f8c…`; run affected/Bash/reference gates. - A prepared patch is cached in functions store, but regenerate from Git if - unavailable. Never edit a frozen run/candidate/receipt. -3. Checkpoint, safely refresh the immutable local installation offline using +1. Both accepted runs and all artifact hashes are verified. Integration and + local affected/Bash/reference gates passed; do not repeat the full suite. + Never edit a frozen run/candidate/receipt. +2. Checkpoint, safely refresh the immutable local installation offline using existing Python 3.12.14 and MCP 2.2.0 wheelhouse; verify idempotence, drift, pip check, doctor, matching host registrations and installed SDK tests. -4. Run **one** bounded Gemini/Antigravity CLI/MCP status recheck against the +3. Run **one** bounded Gemini/Antigravity CLI/MCP status recheck against the final installed launcher and accepted run. Do not repeat completed Grok or Claude host proofs. -5. Record final results/limits in backlog and installed evidence. Whole-plan +4. Record final results/limits in backlog and installed evidence. Whole-plan R3/R4–R7/C1 closure remains separate; use SOL-REVIEW-FOLLOWUP.md next. Only one full core suite may run at a time; freeze source/tests while it runs. @@ -131,109 +131,3 @@ Private bounded helpers are under observe-run.py, inspect-handoff.py, complete-handoff.py, start-combined-review.py, g4-candidate-cases.py and probe.py. Claims/raw logs remain private; never reuse the completed b950 handoff's prior claim for a different run. - Jev pilot, and local Laya fallback plus measured adoption. It prioritizes routing - hints, skill/tool shortlists and context ranking, followed by failure triage, - review attention and outcome labels. No weights/inference/API spending or - runtime routing changes occurred during that original planning pass. The user authorized one Jev request - using only the synthetic fixture, no retries and at most $0.01. The tracked - fixture/probe checkpoint is committed at `70e59cb` and five focused offline - tests are ready; the complete offline - gate is 231 core tests discovered (suite OK, 2 optional SDK skips) and 220 - Bash assertions. The live call was blocked until the October 1 key setup; - it has now run once as recorded above. Classifier suggestions never become permission/acceptance authority. - Do not wait for this probe to execute the review repairs starting at R1. - -## Completed and preserved - -Branch cleanup is complete. Local/GitHub working branch names were consolidated into `main` and `codex/engineering-team`. The old assessment and holdout commits remain in their descendant histories. The unrelated February backup is preserved by its existing local tag and a verified complete Git bundle. See the [branch record](../../audits/2026-09-06-branch-consolidation.md). - -The M1 implementation includes Python packaging/contracts, native Codex protocol preparation and framing, catalog-to-profile preparation, shared error-classification policy, strict input validation, and legacy timeout/context/catalog fixes. Earlier review defects have corresponding regression tests, including [the independent review cases](../../../test/core/test_m1_gate_review.py). - -Verified at the implementation/evidence checkpoints above: - -| Check | Result | -|---|---| -| Python core discovery | 317 tests passed through the M7 live Codex-review repair, with 2 optional-SDK skips and ResourceWarning promoted to error | -| Bash 3.2 regression suite | 11 test files, 227 assertions passed | -| Optional MCP boundary | `mcp==2.2.0` installed/constructed on local Python; Python 3.11 lock resolution; 22 official-SDK focused tests passed | -| M7 installed runtime | Current immutable release `0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef` has no source/plugin/installed drift; `pip check`, idempotent reinstall, four-host unchanged setup and doctor passed | -| M7 normal task entry | Exact-commit review and bounded fix need no hand-written JSON; an offline delivery passed writer/candidate/review/check/source-preservation/handoff gates, and a live installed Codex review passed both checks and host acceptance | -| M7 live surface proof | Terminal start/cancel plus real Gemini/Antigravity and ephemeral Codex `squad_status` calls observed the same run/version | -| M4 local host setup | Stable isolated runtime is registered in all four real local configs; doctor reports ready and a second setup pass was unchanged | -| M4 cross-surface proof | Real Codex read the terminal-started run through MCP; official SDK clients proved identical ledger, fenced claims, completion and disconnect survival; actual Claude handoff remains blocked on login | -| Wheel installation | Fresh external venv resolves packaged assets and applies migrations through schema 8 | -| Earlier live probes | Codex metadata and a separate read-only CLI smoke succeeded | -| Integrated native adapter proof | Passed at `97a10f0`; gpt-5.5/low, read-only, correlated completion and confirmed process-group cleanup | -| M2 crash/race matrix | Real subprocess interruptions plus independent-process start, writer, cancel, import and host-handoff races passed at `ddb6f51` | -| M3 deterministic routing | Strict profiles/policy, aliases, overrides, fallback and typed capacity tests passed at `ca55990` / `1c7b614` | -| M3 frozen review input | Exact OIDs/config hashes, detached worktree, moving-ref stability, dirty-input rejection and source checkout preservation passed at `cd9a881` | -| M3 offline workflow evidence | Strict candidate-bound review/check evaluation and durable separate-worktree execution passed at `a756307` / `97d2c6c` | -| M3 host disposition/reporting | Accept/reject/revise, required-check blocking, retry budgets, stale claims, crash resume and five terminal reports passed at `30cf49e` | -| M3 native Codex reviewer | Public start, exact identity verification, ephemeral read-only structured output, native usage and four provider-fault classes passed offline at `459ff3f` | -| M3 live public review | Passed at `9478796`; gpt-5.5/low found one supported regression, the required check passed, host acceptance terminalized succeeded and five report hashes were retained | -| M3 closeout | Waiting/failure reports, headless leadership, cumulative budgets, live pool fencing, runtime fallbacks and all four independent-audit fixes pass at `1737667` | - -The first two saved-probe invocations failed before `Popen` because of -probe-only path/field defects, so neither launched Codex nor consumed a model -turn. Their private receipts remain under `~/.devsquad/private-probes`. The -probe now uses a dedicated process session, bounded group TERM/KILL cleanup, -an explicit terminal deadline, and retains early notifications for correlation. -The successful run retained separate stderr files of 138,030 and -285,644 bytes, supporting the diagnosis that an undrained stderr pipe caused -the earlier apparent nonresponses. - -The authoritative requirement matrices are [M1-STATUS.md](M1-STATUS.md), -[M2-STATUS.md](M2-STATUS.md), [M3-STATUS.md](M3-STATUS.md), -[M5-STATUS.md](M5-STATUS.md), [M6-STATUS.md](M6-STATUS.md) and -[M7-STATUS.md](M7-STATUS.md). [backlog.json](backlog.json) marks M1/M2 complete, -M3/M5/M6/M7 in progress, M4 blocked on its real Claude handoff, and C1 pending. -The review correction and [Sol follow-up plan](SOL-REVIEW-FOLLOWUP.md) govern -where historical verification is incomplete. Unauthenticated or unsupported -provider paths must not be advertised as verified. - -## Exact next work - -1. Check Git status and recent commits, preserving work newer than this note. - Continue `codex/engineering-team`; do not restart from `main` or redo M1/M2. -2. R3b.1's source/offline gate passed at `a4a87fd`; do not repeat unchanged - tests. Implement R3b.2 shared current-evidence eligibility and append-only - evaluation/review revisions. Validate every relevant attempt, including - fallbacks/repairs. Finish - R3b/R3c saved evidence, replay, - qualification/promotion/rollback/catalog-fallback eligibility and historical - compatibility, then execute R4–R6 in dependency order. Preserve R1/R2, - explicit check `output_paths` contract and historical receipts. Do not - rewrite the architecture or reset completed work. -3. Claude login is now confirmed; the user reports Grok signed in. After the - repaired installation is safely refreshed under R8, run the M4 real Claude - handoff, the M5 installed Claude-to-Codex delivery and one bounded Grok - operation, retaining only redacted evidence. Authentication is not itself - a supported-operation receipt. -4. Keep M6 decision guidance off. The one-request Jev pilot is complete and - must not be repeated under its spent authorization. Preserve its tier - disagreements and frozen labels; a broader shadow/adoption comparison needs - predeclared gates and a separate budget. Install/run Laya only if the - declared cost/access/quality trigger is established. -5. Complete the separately gated C1 extension under R7 and audit installed/live - closure under R8. C1 is required in the full assignment even though it does - not reopen M7. Record each blocked subgate without pausing unrelated work. - -The local official reference clone `/tmp/devsquad-codex-plugin-review-20260906` has native client patterns, including the `initialize` → `initialized` handshake. Installed protocol schemas were generated under `/tmp/devsquad-codex-protocol-20260906`. These temporary references may need to be regenerated after a restart; they are not the project source of truth. - -## Checkpoint discipline - -- Commit coherent partial work and its evidence at small intervals; do not wait for an entire milestone. Mark incomplete work accurately. -- Before long probes or a likely usage cutoff, update this recovery note and checkpoint. Keep the working tree clean at a pause; never stash. -- Run the required `bash test/run.sh` before each commit, and the relevant core tests for code changes. Record failing checks when saving a necessary WIP checkpoint rather than calling it complete. -- Keep raw private prompts, credentials and native diagnostic logs outside tracked evidence. Preserve reproducible scripts and redacted receipts in the repository. -- An account limit does not authorize purchasing credits, consuming a reset credit, silently using paid APIs or changing the requested implementation model. Resume when capacity is available or the user supplies new instructions. -- Do not promise execution while the account is blocked. The committed work and this file are the handoff across that interruption. - -```bash -git status --short --branch -git log -6 --oneline -PYTHONDONTWRITEBYTECODE=1 python3 -m unittest discover -s test/core -v -bash test/run.sh -``` - -Continue from the earliest unfinished requirement with available dependencies. Preserve all later implementation and review findings if this note is older than the current branch. diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index 05dd44c..28b7d4b 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -1,8 +1,9 @@ { "schema_version": 1, - "status": "accepted_477_test_native_workflow_final_combined_review_running", + "status": "accepted_477_test_workflow_combined_review_and_source_integration_passed_install_pending", "fresh_byte_correction": {"run_id":"360a4993-d016-406e-8bb4-ae6d57dfe6cf","state_at_checkpoint":"running","version":4,"target":"af994e948eb38764884d27fd61665efa776cb47b","scope":"two writable files, no revisions, two workers, 1800s wall budget; original frozen task unchanged","mandatory_checks":"diff, Bash and full spawn-safe core suite with explicit offline build interpreter","operator_red":{"case":"real Git index/tree non-UTF-8 documentation and test filenames","result":"both raise UnicodeDecodeError on af994","fixture_note":"Git plumbing inserts byte names into the index; macOS APFS rejects such working-tree names"},"next":"finish candidate/native review/full gates, then narrow full-combined review, integration, offline install and one Gemini recheck","terminal":{"state":"succeeded","version":31,"host":"accept"},"candidate":{"commit":"f8c4f8c83f170870eb37d564e54eee6188fc233c","sha256":"3f0fa0e86edd8a4c7b04bc1a2447d70b136e596cafe3624826f1f46c2fcee568"},"writer":{"model":"claude-sonnet-5","verification":"verified","harness":"2.1.220 (Claude Code)","input_tokens":12,"output_tokens":12079,"total_tokens":12091},"review":{"model":"gpt-6.1-sol","effort":"low","verification":"verified","permission_policy":"read_only","verdict":"clean","input_tokens":90119,"output_tokens":481,"total_tokens":90600},"gates":{"checks":"all_three_passed_unchanged_verified_integrity","bash_assertions":227,"full_core":{"tests":477,"unittest_seconds":455.86,"monotonic_seconds":456.2232173750008,"utc_seconds":456.21989,"errors":0,"failures":0,"skips":2,"unraisable":[],"result":"pass"},"checks_artifact_sha256":"ee7d7c903872364b6c3172d2b68084f8e352578ef666564e58dcbd1a63985820"},"operator_recheck":"All five README/symlink/Unicode/byte-documentation/byte-test cases pass"}, - "byte_fix_combined_review": {"run_id": "a6ecd887-5122-4281-b988-4d344a662e20", "state_at_checkpoint": "running", "version": 4, "base": "1781b89b1228dfca2ca9148df437af3ac04b971f", "target": "f8c4f8c83f170870eb37d564e54eee6188fc233c", "scope": "five changed files", "checks": "mandatory diff, Bash and 27 affected tests; no duplicate full gate"}, + "byte_fix_combined_review": {"run_id":"a6ecd887-5122-4281-b988-4d344a662e20","state_at_checkpoint":"running","version":4,"base":"1781b89b1228dfca2ca9148df437af3ac04b971f","target":"f8c4f8c83f170870eb37d564e54eee6188fc233c","scope":"five changed files","checks":"mandatory diff, Bash and 27 affected tests; no duplicate full gate","terminal":{"state":"succeeded","version":22,"host":"accept"},"review":{"model":"gpt-6.1-sol","effort":"low","verification":"verified","verdict":"clean","input_tokens":76865,"output_tokens":716,"total_tokens":77581},"gates":{"checks":"all_three_passed_with_unchanged_verified_integrity","tests":27,"seconds":24.173,"checks_artifact_sha256":"4bbc33c831fdfeed25f3329dd51ed5e0487648bfdf5ada047d23e3fdd692b30e"}}, + "source_integration": {"candidate":"f8c4f8c83f170870eb37d564e54eee6188fc233c","exact_blob_matches":[{"path":"plugin/core/src/devsquad/task_entry.py","blob":"d350e8409fcb3a47a9d1d10a81dbbc2be184b308","matched":true},{"path":"test/core/test_task_entry.py","blob":"907596e10915f3935c4dbe636f75e044021c6ae5","matched":true},{"path":"test/core/test_cli.py","blob":"8acb256d06a6b0ca8c1cfd8b7be85f7dde6baf82","matched":true},{"path":"test/core/test_handoff_store.py","blob":"2293bfd9ad57c5749b539438191ee62ae4624bff","matched":true},{"path":"test/core/test_mcp.py","blob":"dccf29017f77580c3f7500ff321c88e254052676","matched":true}],"affected":{"tests":27,"seconds":18.94,"result":"pass"},"bash_assertions":227,"reference":"current","diff_check":"pass","full_gate":"accepted exact-candidate 477-test run; not unnecessarily repeated"}, "source_checkpoint": "c7ccc02", "installation": { "previous_release": "0.1.0-py31214-9e5cdea2aa99-mcp-a26bc88afbef", diff --git a/plugin/core/src/devsquad/task_entry.py b/plugin/core/src/devsquad/task_entry.py index 75243d1..d350e84 100644 --- a/plugin/core/src/devsquad/task_entry.py +++ b/plugin/core/src/devsquad/task_entry.py @@ -294,11 +294,87 @@ def _managed_routing( } -def _tracked(repo: Path, relative: str) -> bool: +def _tree_entry(repo: Path, oid: str, relative: str) -> tuple[str, str] | None: + """Return (mode, type) for a path in the exact oid tree, or None if absent.""" + output = _git(repo, "ls-tree", oid, "--", relative) + if not output: + return None + meta, _, _ = output.splitlines()[0].partition("\t") + parts = meta.split() + if len(parts) < 2: + return None + return parts[0], parts[1] + + +def _is_regular_blob(repo: Path, oid: str, relative: str) -> bool: + entry = _tree_entry(repo, oid, relative) + return entry is not None and entry[1] == "blob" and entry[0] != "120000" + + +def _is_tree(repo: Path, oid: str, relative: str) -> bool: + entry = _tree_entry(repo, oid, relative) + return entry is not None and entry[1] == "tree" + + +def _git_entries_z(repo: Path, *arguments: str) -> list[str]: + """Run git and split NUL-delimited output, preserving embedded tabs/newlines.""" try: - return bool(_git(repo, "ls-files", "--error-unmatch", "--", relative)) - except ContractError: + completed = subprocess.run( + ["git", "-C", str(repo), *arguments], + capture_output=True, + timeout=10, + check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise ContractError("cannot inspect the Git project") from exc + if completed.returncode != 0: + raise ContractError(f"Git project check failed: {' '.join(arguments[:2])}") + return [ + entry.decode("utf-8", errors="surrogateescape") + for entry in completed.stdout.split(b"\0") + if entry + ] + + +def _has_python_tests(repo: Path, oid: str, relative: str) -> bool: + """Return True if a real tracked tree at `relative` holds a regular test*.py file.""" + if not _is_tree(repo, oid, relative): return False + for entry in _git_entries_z(repo, "ls-tree", "-r", "-z", oid, "--", relative): + meta, _, path = entry.partition("\t") + parts = meta.split() + if len(parts) < 2: + continue + mode, object_type = parts[0], parts[1] + if object_type != "blob" or mode == "120000": + continue + name = PurePosixPath(path).name + if name.startswith("test") and name.endswith(".py"): + return True + return False + + +def _detect_tests(repo: Path, oid: str) -> list[tuple[str, ...]]: + detected: list[tuple[str, ...]] = [] + if _is_regular_blob(repo, oid, "test/run.sh"): + detected.append(("bash", "test/run.sh")) + elif _has_python_tests(repo, oid, "tests"): + detected.append(("python3", "-m", "unittest", "discover", "-s", "tests")) + elif _has_python_tests(repo, oid, "test"): + detected.append(("python3", "-m", "unittest", "discover", "-s", "test")) + if ( + _is_tree(repo, oid, "plugin/core/src") + and _is_tree(repo, oid, "test/core") + and _is_regular_blob(repo, oid, "scripts/run-core-tests.py") + ): + detected.append(( + "env", + "PYTHONPATH=plugin/core/src:test/core", + "PYTHONWARNINGS=error::ResourceWarning", + "python3", + "scripts/run-core-tests.py", + )) + return detected def _checks( @@ -322,22 +398,22 @@ def _checks( "timeout_seconds": min(timeout_seconds, 120), "required_to_pass": required, }] - detected: tuple[str, ...] | None = None - if _tracked(repo, "test/run.sh"): - detected = ("bash", "test/run.sh") - elif (repo / "tests").is_dir(): - detected = ("python3", "-m", "unittest", "discover", "-s", "tests") - elif (repo / "test").is_dir(): - detected = ("python3", "-m", "unittest", "discover", "-s", "test") - if detected is not None: + seen: set[tuple[str, ...]] = set() + detected_ids = ("detected-tests", "detected-core-tests") + for check_id, argv in zip(detected_ids, _detect_tests(repo, target_oid)): checks.append({ - "id": "detected-tests", - "argv": list(detected), + "id": check_id, + "argv": list(argv), "cwd": ".", "timeout_seconds": timeout_seconds, "required_to_pass": required, }) - for index, arguments in enumerate(supplied, 1): + seen.add(argv) + index = 0 + for arguments in supplied: + if arguments in seen: + continue + index += 1 checks.append({ "id": f"user-check-{index}", "argv": list(arguments), @@ -345,6 +421,7 @@ def _checks( "timeout_seconds": timeout_seconds, "required_to_pass": required, }) + seen.add(arguments) if len(checks) > 16: raise ContractError("normal entry produced too many checks") return checks diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 0869d2b..8acb256 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -661,14 +661,35 @@ def build_python(): shutil.which("python3.11"), ] for candidate in dict.fromkeys(value for value in candidates if value): - result = subprocess.run([ - candidate, "-c", - "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68", - ], text=True, capture_output=True) + try: + result = subprocess.run([ + candidate, "-c", + "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68", + ], text=True, capture_output=True) + except OSError: + continue if result.returncode == 0: return candidate return None + def test_build_python_skips_missing_first_candidate_for_supported_interpreter(self): + missing = str(Path(tempfile.mkdtemp(prefix="devsquad-missing-")) / "no-such-python") + probed = [] + + def fake_run(argv, **kwargs): + probed.append(argv[0]) + if argv[0] == missing: + raise FileNotFoundError(argv[0]) + return subprocess.CompletedProcess(argv, 0) + + with ( + mock.patch.dict(os.environ, {"DEVSQUAD_BUILD_PYTHON": missing}), + mock.patch("subprocess.run", side_effect=fake_run), + ): + result = self.build_python() + self.assertEqual(probed[0], missing) + self.assertEqual(result, sys.executable) + def test_installed_wheel_contains_and_applies_current_migrations(self): build_python = self.build_python() if build_python is None: diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 3a42fba..2293bfd 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -605,19 +605,40 @@ def build_python(): shutil.which("python3.11"), ] for candidate in dict.fromkeys(value for value in candidates if value): - result = subprocess.run( - [ - candidate, - "-c", - "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68", - ], - text=True, - capture_output=True, - ) + try: + result = subprocess.run( + [ + candidate, + "-c", + "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68", + ], + text=True, + capture_output=True, + ) + except OSError: + continue if result.returncode == 0: return candidate return None + def test_build_python_skips_missing_first_candidate_for_supported_interpreter(self): + missing = str(Path(tempfile.mkdtemp(prefix="devsquad-missing-")) / "no-such-python") + probed = [] + + def fake_run(argv, **kwargs): + probed.append(argv[0]) + if argv[0] == missing: + raise FileNotFoundError(argv[0]) + return subprocess.CompletedProcess(argv, 0) + + with ( + mock.patch.dict(os.environ, {"DEVSQUAD_BUILD_PYTHON": missing}), + mock.patch("subprocess.run", side_effect=fake_run), + ): + result = self.build_python() + self.assertEqual(probed[0], missing) + self.assertEqual(result, sys.executable) + def test_installed_wheel_applies_schema_four_to_twelve(self): build_python = self.build_python() if build_python is None: diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index 18f6d32..dccf290 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -788,15 +788,36 @@ def build_python(): shutil.which("python3.11"), ] for candidate in dict.fromkeys(value for value in candidates if value): - result = subprocess.run( - [candidate, "-c", "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68"], - text=True, - capture_output=True, - ) + try: + result = subprocess.run( + [candidate, "-c", "import setuptools, wheel; assert int(setuptools.__version__.split('.')[0]) >= 68"], + text=True, + capture_output=True, + ) + except OSError: + continue if result.returncode == 0: return candidate return None + def test_build_python_skips_missing_first_candidate_for_supported_interpreter(self): + missing = str(Path(tempfile.mkdtemp(prefix="devsquad-missing-")) / "no-such-python") + probed = [] + + def fake_run(argv, **kwargs): + probed.append(argv[0]) + if argv[0] == missing: + raise FileNotFoundError(argv[0]) + return subprocess.CompletedProcess(argv, 0) + + with ( + mock.patch.dict(os.environ, {"DEVSQUAD_BUILD_PYTHON": missing}), + mock.patch("subprocess.run", side_effect=fake_run), + ): + result = self.build_python() + self.assertEqual(probed[0], missing) + self.assertEqual(result, sys.executable) + def test_plain_installed_wheel_keeps_cli_usable_without_mcp(self): build_python = self.build_python() if build_python is None: diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py index 31ee3cf..907596e 100644 --- a/test/core/test_task_entry.py +++ b/test/core/test_task_entry.py @@ -67,6 +67,55 @@ def setUp(self): "effort": "low", } + def _init_repo(self): + temp = tempfile.TemporaryDirectory(prefix="devsquad-task-entry-discovery-") + self.addCleanup(temp.cleanup) + repo = Path(temp.name) / "project" + repo.mkdir() + subprocess.run( + ["git", "init", "-b", "main"], cwd=repo, + check=True, text=True, capture_output=True, + ) + subprocess.run(["git", "config", "user.name", "DevSquad Test"], cwd=repo, check=True) + subprocess.run( + ["git", "config", "user.email", "test@example.invalid"], cwd=repo, check=True, + ) + return repo + + def _commit(self, repo, message): + subprocess.run(["git", "add", "-A"], cwd=repo, check=True) + subprocess.run( + ["git", "commit", "-m", message], cwd=repo, + check=True, text=True, capture_output=True, + ) + return subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=repo, + check=True, text=True, capture_output=True, + ).stdout.strip() + + def _stage_raw_path_blob(self, repo, path_bytes: bytes, content: bytes) -> None: + """Stage a blob at a raw byte path, bypassing filesystem filename rules.""" + hashed = subprocess.run( + ["git", "-C", str(repo), "hash-object", "-w", "--stdin"], + input=content, check=True, capture_output=True, + ) + sha = hashed.stdout.strip().decode() + cacheinfo = f"100644,{sha},".encode() + path_bytes + subprocess.run( + ["git", "-C", str(repo), "update-index", "--add", "--cacheinfo", cacheinfo], + check=True, capture_output=True, + ) + + def _commit_index(self, repo, message): + subprocess.run( + ["git", "-C", str(repo), "commit", "-m", message], + check=True, capture_output=True, + ) + return subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + check=True, text=True, capture_output=True, + ).stdout.strip() + def test_branch_review_freezes_exact_commits_and_embedded_routing(self): task, summary = build_managed_task( workflow="branch-review", @@ -328,6 +377,339 @@ def test_entry_rejects_ambiguous_scope_focus_identity_and_checks(self): with self.assertRaisesRegex(ContractError, "at most 12"): parse_checks(["true"] * 13) + def test_check_discovery_inspects_selected_target_not_current_checkout(self): + repo = self._init_repo() + (repo / "src").mkdir() + (repo / "src/app.py").write_text("VALUE = 1\n") + without_tests = self._commit(repo, "no tests") + (repo / "test").mkdir() + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + with_tests = self._commit(repo, "add bash tests") + + subprocess.run( + ["git", "checkout", without_tests], cwd=repo, + check=True, text=True, capture_output=True, + ) + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=without_tests, target_ref=with_tests, + goal="Target tree has tests though the checkout does not.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual(detected["argv"], ["bash", "test/run.sh"]) + + subprocess.run( + ["git", "checkout", with_tests], cwd=repo, + check=True, text=True, capture_output=True, + ) + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=without_tests, target_ref=without_tests, + goal="Target tree lacks tests though the checkout has them.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + def test_check_discovery_python_tests_follow_selected_target_not_checkout(self): + repo = self._init_repo() + (repo / "src").mkdir() + (repo / "src/app.py").write_text("VALUE = 1\n") + without_tests = self._commit(repo, "no tests tree") + (repo / "tests").mkdir() + (repo / "tests/test_sample.py").write_text("def test_ok():\n assert True\n") + with_tests = self._commit(repo, "add python tests tree") + + subprocess.run( + ["git", "checkout", without_tests], cwd=repo, + check=True, text=True, capture_output=True, + ) + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=without_tests, target_ref=with_tests, + goal="Target tree has Python tests though the checkout does not.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + subprocess.run( + ["git", "checkout", with_tests], cwd=repo, + check=True, text=True, capture_output=True, + ) + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=without_tests, target_ref=without_tests, + goal="Target tree lacks Python tests though the checkout has them.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + def test_check_discovery_selected_target_without_tests(self): + repo = self._init_repo() + (repo / "src").mkdir() + (repo / "src/app.py").write_text("VALUE = 1\n") + oid = self._commit(repo, "no tests at all") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="No tests are tracked anywhere in the selected target.", + codex_identity=self.codex, + ) + self.assertEqual( + [check["id"] for check in task["checks"]], ["candidate-diff-check"], + ) + + def test_check_discovery_detects_real_python_tests(self): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests/test_sample.py").write_text("def test_ok():\n assert True\n") + (repo / "tests/README").write_text("Docs alongside real tests.\n") + oid = self._commit(repo, "real python tests") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="Real tracked test*.py files under tests/ are detected.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + def test_check_discovery_detects_unicode_tab_and_newline_named_python_tests(self): + names = ("test_café.py", "test\tplan.py", "test\nplan.py") + for name in names: + with self.subTest(name=name): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests" / name).write_text("def test_ok():\n assert True\n") + oid = self._commit(repo, "special filename test") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="Unicode, tab, and newline named test*.py files are detected.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + def test_check_discovery_detects_tests_beside_non_utf8_documentation_filename(self): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests/test_sample.py").write_text("def test_ok():\n assert True\n") + subprocess.run(["git", "-C", str(repo), "add", "-A"], check=True, capture_output=True) + self._stage_raw_path_blob( + repo, b"tests/doc_\xff.md", b"Non-UTF-8 named documentation.\n", + ) + oid = self._commit_index(repo, "python tests plus a non-utf8 doc filename") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="A non-UTF-8 documentation filename alongside real tests must not raise.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + def test_check_discovery_detects_non_utf8_named_python_test_file(self): + repo = self._init_repo() + self._stage_raw_path_blob( + repo, b"tests/test_\xff.py", b"def test_ok():\n assert True\n", + ) + oid = self._commit_index(repo, "non-utf8 named python test file") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="A non-UTF-8 named test*.py file must itself be detected.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "tests"], + ) + + def test_check_discovery_excludes_readme_only_and_symlinked_test_trees(self): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests/README").write_text("Not a test suite.\n") + readme_only = self._commit(repo, "readme only tests dir") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=readme_only, target_ref=readme_only, + goal="A tests/README must not be treated as Python tests.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + (repo / "tests/README").unlink() + (repo / "tests").rmdir() + real_dir = Path(self.temp.name) / "external-tests" + real_dir.mkdir() + (real_dir / "test_real.py").write_text("def test_ok():\n assert True\n") + (repo / "tests").symlink_to(real_dir) + symlinked_dir = self._commit(repo, "symlinked tests dir") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=symlinked_dir, target_ref=symlinked_dir, + goal="A symlinked tests directory must not be treated as Python tests.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + def test_check_discovery_excludes_symlinked_test_file(self): + repo = self._init_repo() + (repo / "tests").mkdir() + (repo / "tests/real_test.py").write_text("def test_ok():\n assert True\n") + (repo / "tests/test_link.py").symlink_to("real_test.py") + oid = self._commit(repo, "symlinked python test file") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="A symlinked test*.py file must not be treated as a real Python test.", + codex_identity=self.codex, + ) + self.assertNotIn("detected-tests", [c["id"] for c in task["checks"]]) + + def test_check_discovery_requires_regular_blob_not_symlink(self): + repo = self._init_repo() + (repo / "test").mkdir() + (repo / "test/test_sample.py").write_text("def test_ok():\n assert True\n") + (repo / "test/real.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + (repo / "test/real.sh").chmod(0o755) + (repo / "test/run.sh").symlink_to("real.sh") + symlinked = self._commit(repo, "symlinked run.sh") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=symlinked, target_ref=symlinked, + goal="A symlinked run.sh must not select Bash.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual( + detected["argv"], ["python3", "-m", "unittest", "discover", "-s", "test"], + ) + + (repo / "test/run.sh").unlink() + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + regular = self._commit(repo, "regular run.sh") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=regular, target_ref=regular, + goal="A regular blob run.sh selects Bash.", + codex_identity=self.codex, + ) + detected = next(c for c in task["checks"] if c["id"] == "detected-tests") + self.assertEqual(detected["argv"], ["bash", "test/run.sh"]) + + def test_check_discovery_combines_bash_and_core_runner(self): + repo = self._init_repo() + (repo / "test").mkdir() + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + (repo / "test/core").mkdir() + (repo / "test/core/test_sample.py").write_text("def test_ok():\n assert True\n") + (repo / "plugin/core/src/devsquad").mkdir(parents=True) + (repo / "plugin/core/src/devsquad/__init__.py").write_text("") + (repo / "scripts").mkdir() + (repo / "scripts/run-core-tests.py").write_text("#!/usr/bin/env python3\n") + oid = self._commit(repo, "bash plus core runner") + task, _ = build_managed_task( + workflow="branch-review", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="Both Bash and the DevSquad core runner are detected.", + codex_identity=self.codex, + ) + self.assertEqual( + [ + check["argv"] for check in task["checks"] + if check["id"] in {"detected-tests", "detected-core-tests"} + ], + [ + ["bash", "test/run.sh"], + [ + "env", + "PYTHONPATH=plugin/core/src:test/core", + "PYTHONWARNINGS=error::ResourceWarning", + "python3", + "scripts/run-core-tests.py", + ], + ], + ) + + def test_check_discovery_deduplicates_supplied_argv_matching_detected(self): + repo = self._init_repo() + (repo / "test").mkdir() + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + oid = self._commit(repo, "bash only") + task, _ = build_managed_task( + workflow="issue-delivery", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="An exact supplied duplicate of a detected check runs once.", + codex_identity=self.codex, + checks=parse_checks([ + "bash test/run.sh", + "python3 -m unittest discover -s test", + ]), + ) + self.assertEqual( + [(check["id"], check["argv"]) for check in task["checks"]], + [ + ("candidate-diff-check", ["git", "diff", "--check", oid, "HEAD", "--"]), + ("detected-tests", ["bash", "test/run.sh"]), + ( + "user-check-1", + ["python3", "-m", "unittest", "discover", "-s", "test"], + ), + ], + ) + self.assertTrue(all(check["required_to_pass"] for check in task["checks"])) + + def test_check_discovery_deduplicates_supplied_core_runner_argv(self): + repo = self._init_repo() + (repo / "test/core").mkdir(parents=True) + (repo / "test/core/test_sample.py").write_text("def test_ok():\n assert True\n") + (repo / "plugin/core/src/devsquad").mkdir(parents=True) + (repo / "plugin/core/src/devsquad/__init__.py").write_text("") + (repo / "scripts").mkdir() + (repo / "scripts/run-core-tests.py").write_text("#!/usr/bin/env python3\n") + oid = self._commit(repo, "core runner with python tests tree") + task, _ = build_managed_task( + workflow="issue-delivery", project_dir=repo, + base_ref=oid, target_ref=oid, + goal="A supplied duplicate of the full core runner check runs once.", + codex_identity=self.codex, + checks=parse_checks([ + "env PYTHONPATH=plugin/core/src:test/core " + "PYTHONWARNINGS=error::ResourceWarning python3 scripts/run-core-tests.py", + ]), + ) + self.assertEqual( + [(check["id"], check["argv"]) for check in task["checks"]], + [ + ("candidate-diff-check", ["git", "diff", "--check", oid, "HEAD", "--"]), + ( + "detected-tests", + ["python3", "-m", "unittest", "discover", "-s", "test"], + ), + ( + "detected-core-tests", + [ + "env", + "PYTHONPATH=plugin/core/src:test/core", + "PYTHONWARNINGS=error::ResourceWarning", + "python3", + "scripts/run-core-tests.py", + ], + ), + ], + ) + def test_codex_discovery_selects_requested_exact_model_and_effort(self): manifest = mock.Mock(verified_versions=("codex-cli fixture",)) manifest.resolve_binary.return_value = "/fixture/codex" From 39b95f1c89ca89a73148cd88dfe5565f4d5dc377 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Thu, 1 Oct 2026 18:39:42 -0700 Subject: [PATCH 169/197] WIP checkpoint: Record verified final installation and Gemini recheck; retain open product gates (2026-10-01 18:39) --- docs/plans/engineering-team/RESUME.md | 35 ++++++++++--------- docs/plans/engineering-team/backlog.json | 24 ++++++------- .../R8-installed-workflows-2026-10-01.json | 4 ++- 3 files changed, 33 insertions(+), 30 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index f9d11af..6f90eac 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -8,8 +8,8 @@ remain recoverable in Git; detailed receipts and failed gates stay in evidence. Workspace: /Users/Dikshant/Desktop/Projects/devsquad. Branch: `codex/engineering-team`; never restart this build from main. -Source runtime repairs through `1781b89` are installed. The accepted G4 -candidate below is **integrated, not yet installed**; all five blob hashes +Source runtime repairs and the accepted G4 candidate below are **integrated +and installed at `6d2e0ba`**; all five blob hashes exactly match the accepted candidate. Local affected gate: 27 tests in 18.940s, plus 227 Bash assertions, generated-reference and diff checks passed. @@ -38,17 +38,16 @@ integrity; 27 affected tests in 24.173s. Verified native Codex review is clean. ### Exact next action -1. Both accepted runs and all artifact hashes are verified. Integration and - local affected/Bash/reference gates passed; do not repeat the full suite. - Never edit a frozen run/candidate/receipt. -2. Checkpoint, safely refresh the immutable local installation offline using - existing Python 3.12.14 and MCP 2.2.0 wheelhouse; verify idempotence, drift, - pip check, doctor, matching host registrations and installed SDK tests. -3. Run **one** bounded Gemini/Antigravity CLI/MCP status recheck against the - final installed launcher and accepted run. Do not repeat completed Grok or - Claude host proofs. -4. Record final results/limits in backlog and installed evidence. Whole-plan - R3/R4–R7/C1 closure remains separate; use SOL-REVIEW-FOLLOWUP.md next. +1. The user's three requested runtime actions are verified. Do not repeat + these accepted proofs or the unchanged full suite. Both exact packets and + artifact hashes are verified; never edit frozen evidence. +2. Audit R3b.2/R3c closure against SOL-REVIEW-FOLLOWUP.md and the requirement + matrix, preserving existing reader/eligibility/compatibility repairs. Then + continue R4 catalog/quota, R5 public trials/outcomes, remaining R6 UX and + R7/C1 in dependency order. Whole-plan acceptance is not claimed. +3. Desktop UI proofs remain separate. Antigravity IDE control is permission- + denied; do not bypass it or substitute a CLI receipt. Grok/Gemini MCP status + calls do not prove automatic writer/reviewer adapters. Only one full core suite may run at a time; freeze source/tests while it runs. Use the tracked spawn-safe scripts/run-core-tests.py, never a stdin main. @@ -58,7 +57,7 @@ use git-safety checkpoints, never stash. No goal is currently active. ## Current local installation and host proof Stable launcher: /Users/Dikshant/.local/bin/squad. -Selected release: `0.1.0-py31214-68d542f6e8ea-mcp-a26bc88afbef`. +Selected release: `0.1.0-py31214-01fad439adea-mcp-a26bc88afbef`. Python 3.12.14/MCP 2.2.0, schema 15; previous releases and private pre-upgrade SQLite backup retained. Source/plugin/installed payload drift is false. Upgrade defers for old active/recoverable runs, swaps without migration under @@ -71,10 +70,12 @@ the lock, then lazily migrates with old-client write guards. backed up privately, login/settings preserved. Native grok-4.7-build actually called DevSquad status. This is not automatic Grok writer/reviewer proof. - Antigravity 1.2.13/Gemini 3.8 Flash Low actually called the same MCP status. - Existing project-only grant, plan/sandbox; final-install recheck remains. + Final-install recheck actually observed accepted run 360a4993, succeeded/31, + in 12.118s. Existing project-only status grant, plan/sandbox, no bypass. IDE UI permission was denied; do not bypass it. -- Initial installed SDK gate: 22 passed/no skips; transport follow-up: 9 passed. - Final refreshed-install SDK gate still pending. +- Initial installed SDK gate: 22 passed/no skips; final installed transport + gate: 9 passed in 2.803s, no skips. Reinstall unchanged, no payload drift, + pip check/doctor passed, all four host registrations ready/unchanged. ## Repairs and failure history to preserve diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 7ae5e17..0c982e8 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -17,12 +17,12 @@ "next_work_package": "R3b.2", "partial_implementation_checkpoint": { "recorded_on": "2026-10-01", - "baseline_revision": "a22b949", + "baseline_revision": "6d2e0ba", "status": "partial", - "artifact": "evidence/R3b2-eligibility-partial-2026-10-01.json", - "scope": "Reader full gate passed at a4a87fd; schema-15 explicit revisions and shared lifecycle eligibility implemented with focused public fixtures; no install or provider calls", - "next_action": "Installed schema-15 runtime is refreshed through the tested Claude framing/identity and detached auth repairs. Real Claude host handoff, Grok operation and Gemini CLI/MCP pass. Genuine managed Claude implementation succeeded, but the reviewer died without a receipt. Continue the saved G4 correction run 288ee6f9-f503-421a-b81a-d50ee0cf44d8 through independent review, mandatory tests and fenced acceptance, then integrate and refresh.", - "limitations": "R4 catalog/quota, R5 public controller/outcomes, remaining R6 UX and R7 Council remain open. Managed implementation/review/tests acceptance is pending. Gemini CLI/MCP works; IDE UI permission is denied. Jev remains off; no paid API fallback or reset use authorized." + "artifact": "evidence/R8-installed-workflows-2026-10-01.json", + "scope": "Installed schema-15 runtime through 6d2e0ba; accepted real Claude-to-independent-Codex 477-test delivery, complete combined review and integration, actual Claude handoff/Grok operation and final Gemini CLI/MCP recheck", + "next_action": "Audit R3b.2/R3c closure against SOL-REVIEW-FOLLOWUP, then continue R4 catalog/quota, R5 public trials/outcomes, remaining R6 UX and R7/C1. Do not repeat accepted runtime proofs or unchanged full gates.", + "limitations": "Whole-plan closure is not claimed. Required desktop UI proofs remain unverified; Antigravity IDE permission is denied. Grok/Gemini MCP status does not prove automatic writer/reviewer roles. Jev remains off with its one-request allowance spent; no paid API fallback or reset use authorized." }, "planning_checkpoint": { "recorded_on": "2026-10-01", @@ -47,9 +47,9 @@ {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3b2-eligibility-partial-2026-10-01.json", "checkpoint": "R3a verified at 672e383; R3b.1 reader and failed-output repair passed 427-test full gate at a4a87fd. R3b.2 schema-15 explicit revisions and shared lifecycle eligibility have six new passing tests. Real public offline lifecycle/learning positives replace SQL-terminalized runs without lowering pair gates. Latest 23-test focused gate passes. Stale fallback/rollback, race, revision/CLI and upgraded legacy public proof plus full integration/independent audit remain open. R3 is not closed."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, - {"id": "R6", "title": "Normal terminal experience and readiness", "status": "pending", "milestones": ["M5", "M7"], "depends_on": ["R1", "R2", "R4", "R5"], "items": ["G3", "G4"]}, + {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"G4 discovery verified by accepted native 477-test delivery and combined review; integrated and installed at 6d2e0ba. Remaining UX and R4/R5 dependencies stay open."}, {"id": "R7", "title": "Complete Council within existing runner", "status": "pending", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"]}, - {"id": "R8", "title": "Installed proofs, external gates and closure audit", "status": "pending", "milestones": ["M4", "M5", "M6", "M7", "C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": [], "note": "Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked."} + {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} ], "milestones": [ { @@ -124,7 +124,7 @@ "status": "in_progress", "acceptance_section": "M3 — Ship a useful branch review", "open_review_items": [], - "review_resolution": "F1 source repair is verified under R1; affected installed-runtime refresh remains R8, so M3 is not reclosed yet.", + "review_resolution": "F1/R1 repair is now installed and the complete combined branch review is accepted with all checks passing. Milestone closure audit remains separate; old pending-install wording is superseded by R8-installed-workflows evidence.", "evidence": [ { "kind": "implementation_checkpoint", @@ -245,7 +245,7 @@ "availability": "portable_redacted" } ], - "blocker": "Normal Claude Code provider login is required before the exact real-host handoff acceptance can run; SDK-labeled client evidence is not substituted for that live proof" + "blocker": "Claude authentication and actual CLI/MCP handoff now pass. Required desktop UI proofs remain unverified; Antigravity IDE control is permission-denied. Portable CLI proof is not substituted for desktop local Code-tab proof." }, { "id": "M5", @@ -254,7 +254,7 @@ "status": "in_progress", "acceptance_section": "M5 — Deliver a bounded engineering change", "open_review_items": ["G4"], - "review_resolution": "F1/R1 and F2/R2 source/offline repairs are verified; R6 check coverage and R8 installed/live proof remain open.", + "review_resolution": "F1/R1 and F2/R2 repairs, G4 discovery, installed real Claude-to-independent-Codex 477-test delivery and complete combined review now pass. Whole M5 closure audit and broader R6 scope remain separate.", "evidence": [ { "kind": "implementation_checkpoint", @@ -311,7 +311,7 @@ "availability": "tracked_tests" } ], - "blocker": "R1/R2 source repairs are verified; independent R6 normal-command check coverage remains. The genuine installed Claude-to-Codex delivery gate also requires normal Claude login; fixtures cannot substitute for that receipt." + "blocker": "Claude login, genuine delivery and G4 check coverage are no longer blockers. Remaining acceptance audit and broader R4/R5/R6 dependencies remain open; see installed-workflows evidence." }, { "id": "M6", @@ -490,7 +490,7 @@ "availability": "portable_redacted" } ], - "blocker": "Independent R4/R6 normal-entry, readiness and check-discovery work remains. Final live surface and delivery receipts additionally require Claude login and renewed Grok authentication." + "blocker": "R4 catalog/quota, remaining R6 UX and desktop UI acceptance remain open. Actual Claude delivery/handoff, Grok MCP and Gemini CLI/MCP receipts pass; old login/expired-Grok blockers are superseded." } ], "extensions": [ diff --git a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json index 28b7d4b..971bb7f 100644 --- a/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json +++ b/docs/plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json @@ -1,6 +1,8 @@ { "schema_version": 1, - "status": "accepted_477_test_workflow_combined_review_and_source_integration_passed_install_pending", + "status": "requested_runtime_update_workflows_regressions_and_final_gemini_cli_recheck_passed", + "final_installation": {"source_checkpoint": "6d2e0ba", "release": "0.1.0-py31214-01fad439adea-mcp-a26bc88afbef", "source_digest": "01fad439adea52330b9bcaa7bc61f294ebda6d7a7ccd8ab8e540eefd252767cf", "python": "3.12.14", "mcp": "2.2.0", "network_downloads": false, "nonterminal_runs_before": 0, "first_changed": true, "reinstall_changed": false, "all_payload_drift": false, "manifest_matches": true, "pip_check": "pass", "previous_releases_retained": true, "setup": {"ready": true, "hosts": 4, "unchanged": 4}, "doctor_ready": true, "installed_sdk": {"tests": 9, "seconds": 2.803, "result": "pass", "skips": 0}, "sdk_note": "Printed squad_events argument rejection is the expected negative-contract test, not a failed test"}, + "final_gemini_recheck": {"status": "pass_cli_mcp", "release": "0.1.0-py31214-01fad439adea-mcp-a26bc88afbef", "version": "1.2.13", "model": "gemini-3.8-flash-low", "tool": "devsquad/squad_status", "tool_calls": 1, "tool_response_ok": true, "observed": {"run_id": "360a4993-d016-406e-8bb4-ae6d57dfe6cf", "state": "succeeded", "version": 31}, "returncode": 0, "timed_out": false, "seconds": 12.117595374999837, "stdout_sha256": "32636cfdf65c0695ae389d392da3855ce1d3df35c404d559f4a61b99eae5627f", "stderr_sha256": "3a7f8550106d86abb322e8ed0fb30a0296c0e97c6be03d5a8033b99d16737459", "usage": {"input_tokens": 35090, "output_tokens": 213, "thinking_tokens": 0, "cache_read_tokens": 44857, "total_tokens": 35303}, "permissions": "existing project-only status grant, plan mode, sandbox; no bypass", "schema_read": "only DevSquad squad_status tool schema", "ide_ui": "unverified_permission_denied_do_not_bypass"}, "fresh_byte_correction": {"run_id":"360a4993-d016-406e-8bb4-ae6d57dfe6cf","state_at_checkpoint":"running","version":4,"target":"af994e948eb38764884d27fd61665efa776cb47b","scope":"two writable files, no revisions, two workers, 1800s wall budget; original frozen task unchanged","mandatory_checks":"diff, Bash and full spawn-safe core suite with explicit offline build interpreter","operator_red":{"case":"real Git index/tree non-UTF-8 documentation and test filenames","result":"both raise UnicodeDecodeError on af994","fixture_note":"Git plumbing inserts byte names into the index; macOS APFS rejects such working-tree names"},"next":"finish candidate/native review/full gates, then narrow full-combined review, integration, offline install and one Gemini recheck","terminal":{"state":"succeeded","version":31,"host":"accept"},"candidate":{"commit":"f8c4f8c83f170870eb37d564e54eee6188fc233c","sha256":"3f0fa0e86edd8a4c7b04bc1a2447d70b136e596cafe3624826f1f46c2fcee568"},"writer":{"model":"claude-sonnet-5","verification":"verified","harness":"2.1.220 (Claude Code)","input_tokens":12,"output_tokens":12079,"total_tokens":12091},"review":{"model":"gpt-6.1-sol","effort":"low","verification":"verified","permission_policy":"read_only","verdict":"clean","input_tokens":90119,"output_tokens":481,"total_tokens":90600},"gates":{"checks":"all_three_passed_unchanged_verified_integrity","bash_assertions":227,"full_core":{"tests":477,"unittest_seconds":455.86,"monotonic_seconds":456.2232173750008,"utc_seconds":456.21989,"errors":0,"failures":0,"skips":2,"unraisable":[],"result":"pass"},"checks_artifact_sha256":"ee7d7c903872364b6c3172d2b68084f8e352578ef666564e58dcbd1a63985820"},"operator_recheck":"All five README/symlink/Unicode/byte-documentation/byte-test cases pass"}, "byte_fix_combined_review": {"run_id":"a6ecd887-5122-4281-b988-4d344a662e20","state_at_checkpoint":"running","version":4,"base":"1781b89b1228dfca2ca9148df437af3ac04b971f","target":"f8c4f8c83f170870eb37d564e54eee6188fc233c","scope":"five changed files","checks":"mandatory diff, Bash and 27 affected tests; no duplicate full gate","terminal":{"state":"succeeded","version":22,"host":"accept"},"review":{"model":"gpt-6.1-sol","effort":"low","verification":"verified","verdict":"clean","input_tokens":76865,"output_tokens":716,"total_tokens":77581},"gates":{"checks":"all_three_passed_with_unchanged_verified_integrity","tests":27,"seconds":24.173,"checks_artifact_sha256":"4bbc33c831fdfeed25f3329dd51ed5e0487648bfdf5ada047d23e3fdd692b30e"}}, "source_integration": {"candidate":"f8c4f8c83f170870eb37d564e54eee6188fc233c","exact_blob_matches":[{"path":"plugin/core/src/devsquad/task_entry.py","blob":"d350e8409fcb3a47a9d1d10a81dbbc2be184b308","matched":true},{"path":"test/core/test_task_entry.py","blob":"907596e10915f3935c4dbe636f75e044021c6ae5","matched":true},{"path":"test/core/test_cli.py","blob":"8acb256d06a6b0ca8c1cfd8b7be85f7dde6baf82","matched":true},{"path":"test/core/test_handoff_store.py","blob":"2293bfd9ad57c5749b539438191ee62ae4624bff","matched":true},{"path":"test/core/test_mcp.py","blob":"dccf29017f77580c3f7500ff321c88e254052676","matched":true}],"affected":{"tests":27,"seconds":18.94,"result":"pass"},"bash_assertions":227,"reference":"current","diff_check":"pass","full_gate":"accepted exact-candidate 477-test run; not unnecessarily repeated"}, From 9dfa74c11e5043155868ffc08e45170fbee6974b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 13:08:56 -0700 Subject: [PATCH 170/197] WIP checkpoint: WIP R4: scoped native catalog and quota integration; R3 closure audit next (2026-10-02 13:08) --- docs/plans/engineering-team/RESUME.md | 17 ++- plugin/core/src/devsquad/cli.py | 1 + plugin/core/src/devsquad/codex_protocol.py | 6 +- plugin/core/src/devsquad/native_catalog.py | 152 +++++++++++++++++++++ plugin/core/src/devsquad/task_entry.py | 55 ++++++-- test/core/test_native_catalog.py | 101 ++++++++++++++ test/core/test_task_entry.py | 52 ++++++- 7 files changed, 373 insertions(+), 11 deletions(-) create mode 100644 plugin/core/src/devsquad/native_catalog.py create mode 100644 test/core/test_native_catalog.py diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 6f90eac..2ad77be 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -4,7 +4,22 @@ This is the authoritative recovery entry point, not a chronological chat log. Compare it with Git status/recent commits and retain newer work. Earlier notes remain recoverable in Git; detailed receipts and failed gates stay in evidence. -## Current checkpoint — October 1, 2026 +## Current checkpoint — October 2, 2026 + +Continuation: R3c final independent closure audit is next, against immutable +`39b95f1` and the existing 477-test full gate. No required R3/R4 acceptance is +being silently waived. A coherent **partial R4** source slice is checkpointed: +native normal commands now call account/config/model/quota read-only RPCs; +scoped private last-good catalogs use a 24-hour TTL, OS refresh lease and +two-minute failure backoff. Native windows enter the shared typed capacity +ledger. Alias account/config/catalog mismatches block for requalification; +explicit pins remain trials. No login, reset-credit or API fallback calls. +New cache/concurrency/privacy/window and two-project discovery tests are added. +Focused/full/independent package closure and fresh-install proof are still +required before R4 is closed. One real non-generating native discovery probe +verified gpt-6.1-sol/low, scoped pool/catalog fingerprint and two quota windows; +reported capacity was unknown (not guessed available). Continue R3c audit, +then R4 public preflight/lease/identity acceptance, R5, R6 and R7/C1. Workspace: /Users/Dikshant/Desktop/Projects/devsquad. Branch: `codex/engineering-team`; never restart this build from main. diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 490028e..a9a7efd 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -230,6 +230,7 @@ def _command_normal_entry( repo, requested_model=reviewer_model or incumbent.get("model_id"), requested_effort=reviewer_effort or incumbent.get("effort", {}).get("value"), + runtime=Path(args.runtime_dir), ) if workflow == "branch-review": mode = args.mode diff --git a/plugin/core/src/devsquad/codex_protocol.py b/plugin/core/src/devsquad/codex_protocol.py index 2d17210..854b6d2 100644 --- a/plugin/core/src/devsquad/codex_protocol.py +++ b/plugin/core/src/devsquad/codex_protocol.py @@ -371,10 +371,14 @@ def receive_response(peer: JsonLinePeer, request_id: int, *, timeout_seconds: fl def discover_models(peer: JsonLinePeer, *, first_request_id: int = 10, timeout_seconds: float = 5, max_pages: int = 100) -> list[dict[str, Any]]: """Collect a complete native snapshot from an already initialized peer.""" request_id, cursor = first_request_id, None + deadline = time.monotonic() + timeout_seconds responses: list[dict[str, Any]] = [] for _ in range(max_pages): peer.send(model_list_request(request_id, cursor)) - response = receive_response(peer, request_id, timeout_seconds=timeout_seconds) + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("model/list discovery deadline expired") + response = receive_response(peer, request_id, timeout_seconds=remaining) responses.append(response) _, cursor = parse_model_page(response) if cursor is None: diff --git a/plugin/core/src/devsquad/native_catalog.py b/plugin/core/src/devsquad/native_catalog.py new file mode 100644 index 0000000..e590afc --- /dev/null +++ b/plugin/core/src/devsquad/native_catalog.py @@ -0,0 +1,152 @@ +"""Private, scoped last-good discovery and native subscription observations. + +No inference, login, billing, credit-reset, or configuration mutation lives here. +The short-lived OS lock is the refresh lease: process death releases ownership. +""" +from __future__ import annotations + +from datetime import datetime, timedelta, timezone +import fcntl +import hashlib +import json +import math +import os +from pathlib import Path +import uuid +from typing import Any, Callable + +from .catalog import analyze_catalog_drift, normalize_models +from .contracts import ContractError +from .store import canonical_json + +CATALOG_TTL = timedelta(hours=24) +REFRESH_BACKOFF = timedelta(minutes=2) +QUOTA_TTL = timedelta(seconds=60) + + +def native_scope(account_result: dict[str, Any], config: dict[str, Any], binary: str, version: str) -> str: + account = account_result.get("account") + if not isinstance(account, dict) or account.get("type") != "chatgpt": + raise ContractError("normal entry requires native ChatGPT subscription authentication") + identity = {key: account.get(key) for key in ("type", "id", "accountId", "email", "planType")} + if not any(isinstance(identity[key], str) and identity[key] for key in ("id", "accountId", "email")): + raise ContractError("native account identity is unknown; discovery cannot be reused") + # Hash in memory only. Neither account identifiers nor effective configuration + # (which can contain sensitive provider fields) are persisted or displayed. + return hashlib.sha256(canonical_json({ + "account": identity, "config": config, "binary": binary, "version": version, + }).encode()).hexdigest() + + +def normalize_codex_limits(result: dict[str, Any], pool_id: str, *, now: datetime | None = None) -> list[dict[str, Any]]: + current = now or datetime.now(timezone.utc) + buckets = result.get("rateLimitsByLimitId") + bucket = buckets.get("codex") if isinstance(buckets, dict) else result.get("rateLimits") + bucket = bucket if isinstance(bucket, dict) else {} + observations = [] + for slot in ("primary", "secondary"): + window = bucket.get(slot) + window = window if isinstance(window, dict) else {} + used, duration, reset = (window.get(key) for key in ("usedPercent", "windowDurationMins", "resetsAt")) + valid = (type(used) in (int, float) and math.isfinite(used) and 0 <= used <= 100 + and type(duration) is int and duration > 0 and type(reset) is int) + resets_at = None + if valid: + try: + resets_at = datetime.fromtimestamp(reset, timezone.utc) + valid = resets_at > current + except (ValueError, OverflowError, OSError): + valid = False + expires = min(current + QUOTA_TTL, resets_at) if valid else current + QUOTA_TTL + evidence = { + "schema_version": 1, "pool_id": pool_id, "window_id": f"codex:{slot}", + "applies_to": {"harnesses": ["codex"], "model_families": [], "model_ids": []}, + "observed_at": current.isoformat(), "expires_at": expires.isoformat(), + "source": "native_reported", "used": used if valid else None, + "limit": 100 if valid else None, "unit": "percent", + "resets_at": resets_at.isoformat() if valid else None, + "confidence": "reported", + } + evidence["observation_id"] = "native-" + hashlib.sha256(canonical_json(evidence).encode()).hexdigest() + observations.append(evidence) + return observations + + +class NativeCatalogCache: + def __init__(self, directory: Path, scope: str, version: str): + self.directory, self.scope, self.version = directory, scope, version + self.path = directory / (hashlib.sha256(scope.encode()).hexdigest() + ".json") + + def _read(self) -> dict[str, Any] | None: + if not self.path.exists(): + return None + try: + value = json.loads(self.path.read_bytes()) + if (value["scope"] != self.scope or value["harness_version"] != self.version + or value["complete"] is not True or not isinstance(value["models"], list)): + raise ValueError() + return value + except (ValueError, KeyError, TypeError) as exc: + raise ContractError("native catalog cache integrity is invalid") from exc + + @staticmethod + def _recent(timestamp: str, current: datetime, ttl: timedelta) -> bool: + try: + at = datetime.fromisoformat(timestamp) + return at.tzinfo is not None and timedelta(0) <= current - at < ttl + except (TypeError, ValueError): + return False + + def _write(self, value: dict[str, Any]) -> None: + temporary = self.path.with_suffix(f".tmp-{uuid.uuid4().hex}") + try: + with temporary.open("x", encoding="utf-8") as output: + os.chmod(temporary, 0o600) + output.write(canonical_json(value) + "\n") + output.flush() + os.fsync(output.fileno()) + os.replace(temporary, self.path) + finally: + temporary.unlink(missing_ok=True) + + def refresh(self, fetch: Callable[[], list[dict[str, Any]]], *, now: datetime | None = None) -> dict[str, Any]: + current = now or datetime.now(timezone.utc) + self.directory.mkdir(parents=True, exist_ok=True) + with self.path.with_suffix(".lock").open("a+") as lease: + os.chmod(lease.name, 0o600) + try: + fcntl.flock(lease, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError: + previous = self._read() + if previous is None: + raise ContractError("native catalog refresh is in progress; retry shortly") + return previous + try: + previous = self._read() + if previous is not None and ( + self._recent(previous["fetched_at"], current, CATALOG_TTL) + or self._recent(previous["last_refresh"]["at"], current, REFRESH_BACKOFF) + ): + return previous + try: + raw_models = fetch() + if not isinstance(raw_models, list): + raise ContractError("native catalog response is incomplete") + models = normalize_models("codex", self.version, raw_models) + except (ContractError, EOFError, OSError, TimeoutError): + if previous is None: + raise ContractError("native discovery failed with no scoped last-good catalog") from None + previous["last_refresh"] = {"at": current.isoformat(), "status": "error", "error": "discovery_failed"} + self._write(previous) + return previous + value = { + "schema_version": 1, "scope": self.scope, "harness": "codex", + "harness_version": self.version, "fetched_at": current.isoformat(), + "complete": True, "models": models, + "last_refresh": {"at": current.isoformat(), "status": "ok", "error": None}, + } + value["catalog_change"] = analyze_catalog_drift(previous, value) + self._write(value) + return value + finally: + fcntl.flock(lease, fcntl.LOCK_UN) diff --git a/plugin/core/src/devsquad/task_entry.py b/plugin/core/src/devsquad/task_entry.py index d350e84..19f6a83 100644 --- a/plugin/core/src/devsquad/task_entry.py +++ b/plugin/core/src/devsquad/task_entry.py @@ -8,6 +8,7 @@ import shlex import subprocess import sys +import time from typing import Any, Iterable from .adapters import AdapterManifest, harness_version @@ -18,6 +19,7 @@ initialize_request, initialized_notification, receive_response, + request, ) from .contracts import ContractError from .store import canonical_json @@ -85,6 +87,7 @@ def discover_codex_identity( requested_model: str | None = None, requested_effort: str | None = None, timeout_seconds: int = 15, + runtime: Path | None = None, ) -> dict[str, str]: """Discover one currently available exact Codex model without generating.""" @@ -116,13 +119,42 @@ def discover_codex_identity( if "error" in initialized: raise ContractError("Codex native initialization failed") peer.send(initialized_notification()) - models = normalize_models( - "codex", - version, - discover_models( - peer, first_request_id=10, timeout_seconds=timeout_seconds, - ), - ) + scope = None + if runtime is None: + models = normalize_models( + "codex", version, + discover_models(peer, first_request_id=10, timeout_seconds=timeout_seconds), + ) + else: + from .native_catalog import NativeCatalogCache, native_scope, normalize_codex_limits + from .service import Service + + deadline = time.monotonic() + timeout_seconds + def read_native(request_id, method, params=None): + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("native discovery deadline expired") + peer.send(request(request_id, method, params)) + reply = receive_response(peer, request_id, timeout_seconds=remaining) + if "error" in reply or not isinstance(reply.get("result"), dict): + raise ContractError(f"native {method} is unavailable") + return reply["result"] + + account = read_native(2, "account/read", {"refreshToken": False}) + configuration = read_native(3, "config/read", {"includeLayers": False}) + scope = native_scope(account, configuration, binary, version) + catalog = NativeCatalogCache(runtime / "catalogs", scope, version).refresh( + lambda: discover_models(peer, first_request_id=10, timeout_seconds=max(0.001, deadline - time.monotonic())), + ) + models = catalog["models"] + pool_id = f"codex-subscription-{scope}" + try: + limits = read_native(1000, "account/rateLimits/read") + except (ContractError, EOFError, OSError, TimeoutError): + limits = {} + service = Service(runtime) + for observation in normalize_codex_limits(limits, pool_id): + service.capacity_observe(observation) except (EOFError, OSError, TimeoutError) as exc: raise ContractError("Codex model discovery did not complete") from exc finally: @@ -159,6 +191,7 @@ def discover_codex_identity( "model_id": selected["id"], "model_family": family if isinstance(family, str) and family else "gpt", "effort": effort, + **({"account_pool_id": pool_id, "catalog_fingerprint": selected["fingerprint"]} if scope is not None else {}), } @@ -211,6 +244,7 @@ def _managed_routing( "quality_status": "trial", "evidence_refs": [ f"runtime-catalog:{codex['harness_version']}:{codex['model_id']}", + *([f"runtime-catalog-fingerprint:{codex['catalog_fingerprint']}"] if "catalog_fingerprint" in codex else []), ], } trial_profiles = {"reviewer": reviewer} @@ -252,6 +286,11 @@ def _managed_routing( or incumbent["billing_mode"] != "subscription" or incumbent["quality_status"] != "proven"): raise ContractError("normal role binding exceeds the supported role contract") + if role == "reviewer" and "catalog_fingerprint" in codex and ( + incumbent["account_pool_id"] != codex["account_pool_id"] + or f"runtime-catalog-fingerprint:{codex['catalog_fingerprint']}" not in incumbent["evidence_refs"] + ): + raise ContractError("approved reviewer requires requalification for the current native account/config/catalog") if incumbent["id"] == profile["id"] and incumbent != profile: raise ContractError("normal role binding conflicts with the trial profile") if incumbent["id"] != profile["id"]: @@ -465,7 +504,7 @@ def build_managed_task( "harness", "harness_version", "model_id", "model_family", "effort", } if (not isinstance(codex_identity, dict) - or set(codex_identity) - (required_identity | {"account_pool_id"}) + or set(codex_identity) - (required_identity | {"account_pool_id", "catalog_fingerprint"}) or required_identity - set(codex_identity) or codex_identity.get("harness") != "codex" or not all( diff --git a/test/core/test_native_catalog.py b/test/core/test_native_catalog.py new file mode 100644 index 0000000..88f7ab8 --- /dev/null +++ b/test/core/test_native_catalog.py @@ -0,0 +1,101 @@ +import json +from datetime import datetime, timedelta, timezone +from pathlib import Path +import sys +import tempfile +import threading +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.native_catalog import NativeCatalogCache, native_scope, normalize_codex_limits +from devsquad.capacity import derive_pool_capacity + + +NOW = datetime(2026, 10, 2, tzinfo=timezone.utc) +MODELS = [{"id": "gpt-test", "supportedReasoningEfforts": ["low"]}] +TARGET = {"harness": "codex", "model_family": "gpt", "model_id": "gpt-test"} + + +class NativeCatalogTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.cache = NativeCatalogCache(Path(self.temp.name), "scope-a", "v-test") + + def test_last_good_ttl_failure_backoff_and_drift(self): + first = self.cache.refresh(lambda: MODELS, now=NOW) + self.assertTrue(first["complete"]) + self.assertEqual(self.cache.refresh(lambda: self.fail("fresh cache queried"), now=NOW), first) + def failure(): + raise ContractError("AUTH_ERROR private provider diagnostics") + later = NOW + timedelta(days=2) + failed = self.cache.refresh(failure, now=later) + self.assertEqual(failed["models"], first["models"]) + self.assertEqual(failed["last_refresh"]["status"], "error") + self.assertNotIn("private provider", json.dumps(failed)) + self.cache.refresh(lambda: self.fail("backoff ignored"), now=later + timedelta(seconds=10)) + changed = self.cache.refresh(lambda: [], now=later + timedelta(minutes=3)) + self.assertEqual(changed["catalog_change"]["removed_model_ids"], ["gpt-test"]) + + def test_concurrent_refresh_and_dead_owner_release(self): + self.cache.refresh(lambda: MODELS, now=NOW) + entered, release = threading.Event(), threading.Event() + result = [] + def slow(): + entered.set() + self.assertTrue(release.wait(5)) + return MODELS + owner = threading.Thread(target=lambda: result.append(self.cache.refresh(slow, now=NOW + timedelta(days=2)))) + owner.start() + self.assertTrue(entered.wait(5)) + try: + cached = self.cache.refresh(lambda: self.fail("second refresh owner"), now=NOW + timedelta(days=2)) + self.assertEqual(cached["models"][0]["id"], "gpt-test") + finally: + release.set() + owner.join(5) + self.assertEqual(len(result), 1) + def interrupted(): + raise KeyboardInterrupt() + with self.assertRaises(KeyboardInterrupt): + self.cache.refresh(interrupted, now=NOW + timedelta(days=4)) + self.assertTrue(self.cache.refresh(lambda: MODELS, now=NOW + timedelta(days=4))["complete"]) + + def test_scope_is_private_and_changes_for_account_config_binary_version(self): + account = {"account": {"type": "chatgpt", "email": "private@example.invalid", "planType": "plus"}} + scope = native_scope(account, {"provider": "native"}, "/binary/a", "v1") + for a, c, b, v in ( + ({"account": {"type": "chatgpt", "email": "other@example.invalid"}}, {}, "/binary/a", "v1"), + (account, {"provider": "changed"}, "/binary/a", "v1"), + (account, {"provider": "native"}, "/binary/b", "v1"), + (account, {"provider": "native"}, "/binary/a", "v2"), + ): + self.assertNotEqual(scope, native_scope(a, c, b, v)) + self.assertNotIn("private", scope) + for account in ({"account": None}, {"account": {"type": "apiKey"}}, {"account": {"type": "chatgpt"}}): + with self.assertRaises(ContractError): + native_scope(account, {}, "/binary/a", "v1") + other = NativeCatalogCache(Path(self.temp.name), "scope-b", "v-test") + with self.assertRaises(ContractError): + other.refresh(lambda: (_ for _ in ()).throw(TimeoutError()), now=NOW) + + def test_weekly_limit_blocks_available_primary_and_null_is_unknown(self): + payload = {"rateLimitsByLimitId": {"codex": { + "primary": {"usedPercent": 10, "windowDurationMins": 300, "resetsAt": int((NOW + timedelta(hours=5)).timestamp())}, + "secondary": {"usedPercent": 100, "windowDurationMins": 10080, "resetsAt": int((NOW + timedelta(days=5)).timestamp())}, + }}} + observations = normalize_codex_limits(payload, "pool", now=NOW) + self.assertEqual(derive_pool_capacity("pool", observations, target=TARGET, now=NOW)["status"], "exhausted") + self.assertEqual(derive_pool_capacity("pool", observations, target=TARGET, now=NOW + timedelta(minutes=2))["status"], "unknown") + unknown = normalize_codex_limits({"rateLimitsByLimitId": {}, "rateLimits": payload["rateLimitsByLimitId"]["codex"]}, "pool", now=NOW) + self.assertEqual(derive_pool_capacity("pool", unknown, target=TARGET, now=NOW)["status"], "unknown") + self.assertTrue(all(o["used"] is None for o in unknown)) + malformed = normalize_codex_limits({"rateLimits": {"primary": {"usedPercent": True}}}, "pool", now=NOW) + self.assertEqual(derive_pool_capacity("pool", malformed, target=TARGET, now=NOW)["status"], "unknown") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py index 907596e..95a6f80 100644 --- a/test/core/test_task_entry.py +++ b/test/core/test_task_entry.py @@ -303,7 +303,7 @@ def test_public_promotion_changes_normal_entry_but_not_the_old_run_or_pin(self): planned = json.loads(output.getvalue())["data"]["planned_roles"]["reviewer"] self.assertEqual(planned["model_id"], "gpt-new-default" if pin else "gpt-b") self.assertEqual(planned["selection_mode"], "pinned" if pin else "approved_alias") - discovery.assert_called_once_with(self.repo.resolve(), requested_model=planned["model_id"], requested_effort="low") + discovery.assert_called_once_with(self.repo.resolve(), requested_model=planned["model_id"], requested_effort="low", runtime=service.runtime) self.assertEqual(service.normal_entry_bindings("branch-review")["reviewer"]["version"], 8) def test_managed_delivery_runs_offline_through_review_checks_and_handoff(self): @@ -767,6 +767,56 @@ def test_codex_discovery_selects_requested_exact_model_and_effort(self): requested_effort="ultra", ) + def test_normal_native_discovery_caches_across_projects_and_ingests_shared_quota(self): + from datetime import datetime, timedelta, timezone + manifest = mock.Mock(verified_versions=("codex-cli fixture",)) + manifest.resolve_binary.return_value = "/fixture/codex" + process = mock.Mock(stdin=io.StringIO(), stdout=io.StringIO()) + process.poll.return_value = None + runtime = Path(self.temp.name) / "runtime" + second_repo = self._init_repo() + reset = int((datetime.now(timezone.utc) + timedelta(days=2)).timestamp()) + account = {"type": "chatgpt", "email": "fixture@example.invalid", "planType": "plus"} + def reply(peer, request_id, **unused): + return {"result": {1: {}, 2: {"account": account}, 3: {"config": {"provider": "native"}}, + 1000: {"rateLimits": {"primary": {"usedPercent": 5, "windowDurationMins": 300, "resetsAt": reset}, + "secondary": {"usedPercent": 100, "windowDurationMins": 10080, "resetsAt": reset}}}}[request_id]} + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process), + mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), + mock.patch("devsquad.task_entry.receive_response", side_effect=reply), + mock.patch("devsquad.task_entry.discover_models", return_value=[{"id": "gpt-fixture", "supportedReasoningEfforts": ["low"]}]) as discovery, + ): + first = discover_codex_identity(self.repo, runtime=runtime) + second = discover_codex_identity(second_repo, runtime=runtime) + discovery.assert_called_once() + self.assertEqual(first, second) + service = Service(runtime) + store = service._store() + try: + target = {key: first[key] for key in ("harness", "model_family", "model_id")} + self.assertEqual(store.capacity_snapshot(first["account_pool_id"], target=target)["status"], "exhausted") + finally: + store.close() + account["email"] = "another@example.invalid" + other = discover_codex_identity(self.repo, runtime=runtime) + self.assertNotEqual(first["account_pool_id"], other["account_pool_id"]) + self.assertEqual(discovery.call_count, 2) + self.assertNotIn("fixture@example.invalid", "".join(path.read_text() for path in (runtime / "catalogs").glob("*.json"))) + + def test_normal_alias_rejects_account_or_catalog_change_without_mutating_binding(self): + identity = {**self.codex, "account_pool_id": "native-scope", "catalog_fingerprint": "a" * 64} + task, _ = build_managed_task(workflow="branch-review", project_dir=self.repo, base_ref="HEAD", target_ref="HEAD", goal="Bounded review", codex_identity=identity) + incumbent = copy.deepcopy(task["routing"]["profiles"]["profiles"][0]) + incumbent["quality_status"] = "proven" + binding = {"alias": "review.deep", "version": 7, "profile": incumbent} + for change in ({"account_pool_id": "other-scope"}, {"catalog_fingerprint": "b" * 64}): + with self.assertRaisesRegex(ContractError, "requalification"): + build_managed_task(workflow="branch-review", project_dir=self.repo, base_ref="HEAD", target_ref="HEAD", goal="Bounded review", codex_identity={**identity, **change}, role_bindings={"reviewer": binding}) + self.assertEqual(binding["version"], 7) + if __name__ == "__main__": unittest.main() From 500d1b1917ecbcf417af73fbfd21ae67accc45e7 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 13:18:23 -0700 Subject: [PATCH 171/197] WIP checkpoint: R3 closure accepted; R4 scoped discovery and quota public regressions (2026-10-02 13:18) --- docs/plans/engineering-team/RESUME.md | 19 +++++--- docs/plans/engineering-team/backlog.json | 4 +- .../evidence/R3-closure-2026-10-02.json | 44 +++++++++++++++++++ .../R4-native-scoped-partial-2026-10-02.json | 32 ++++++++++++++ plugin/core/src/devsquad/catalog.py | 5 ++- plugin/core/src/devsquad/native_catalog.py | 18 ++++++-- plugin/core/src/devsquad/task_entry.py | 6 ++- test/core/test_native_catalog.py | 8 ++++ test/core/test_task_entry.py | 31 ++++++++++++- 9 files changed, 153 insertions(+), 14 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R3-closure-2026-10-02.json create mode 100644 docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 2ad77be..a2627a4 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,20 +6,29 @@ remain recoverable in Git; detailed receipts and failed gates stay in evidence. ## Current checkpoint — October 2, 2026 -Continuation: R3c final independent closure audit is next, against immutable -`39b95f1` and the existing 477-test full gate. No required R3/R4 acceptance is -being silently waived. A coherent **partial R4** source slice is checkpointed: +R3c is now **accepted** at immutable `39b95f1`: native independent audit is +clean; mandatory diff/Bash/51 focused tests pass with unchanged integrity. +The R3 source blobs exactly match the prior accepted 477-test full candidate. +See `evidence/R3-closure-2026-10-02.json` for the requirement/proof matrix. +A coherent **partial R4** source slice is checkpointed: native normal commands now call account/config/model/quota read-only RPCs; scoped private last-good catalogs use a 24-hour TTL, OS refresh lease and two-minute failure backoff. Native windows enter the shared typed capacity ledger. Alias account/config/catalog mismatches block for requalification; explicit pins remain trials. No login, reset-credit or API fallback calls. New cache/concurrency/privacy/window and two-project discovery tests are added. +Latest stable-source focused gate: **79 tests passed in 14.696s**. Public CLI +entry for two committed projects proves fresh weekly exhaustion prevents any +worker launch; failed quota queries preserve still-fresh prior exhaustion. +R3c independent saved review **`f0a4f86b-b305-4839-88a2-cb5fed8739ab`** is +**succeeded/version 22**, accepted against immutable `39b95f1`; all mandatory +checks and artifact hashes are verified. No active native audit remains. +R4 partial evidence: `evidence/R4-native-scoped-partial-2026-10-02.json`. Focused/full/independent package closure and fresh-install proof are still required before R4 is closed. One real non-generating native discovery probe verified gpt-6.1-sol/low, scoped pool/catalog fingerprint and two quota windows; -reported capacity was unknown (not guessed available). Continue R3c audit, -then R4 public preflight/lease/identity acceptance, R5, R6 and R7/C1. +reported capacity was unknown (not guessed available). Continue R4 full and +independent package/installed acceptance, then R5, R6 and R7/C1. Workspace: /Users/Dikshant/Desktop/Projects/devsquad. Branch: `codex/engineering-team`; never restart this build from main. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 0c982e8..3961d33 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -44,8 +44,8 @@ "review_work_packages": [ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, - {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "in_progress", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3b2-eligibility-partial-2026-10-01.json", "checkpoint": "R3a verified at 672e383; R3b.1 reader and failed-output repair passed 427-test full gate at a4a87fd. R3b.2 schema-15 explicit revisions and shared lifecycle eligibility have six new passing tests. Real public offline lifecycle/learning positives replace SQL-terminalized runs without lowering pair gates. Latest 23-test focused gate passes. Stale fallback/rollback, race, revision/CLI and upgraded legacy public proof plus full integration/independent audit remain open. R3 is not closed."}, - {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "pending", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"]}, + {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, + {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "in_progress", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-native-scoped-partial-2026-10-02.json", "checkpoint": "Partial scoped native cache/quota production connection passes 79 focused tests and a non-generating native probe. Two-project public CLI quota fence verified. R3 independent closure, full integration, independent R4 audit and installed recheck still required; not accepted."}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"G4 discovery verified by accepted native 477-test delivery and combined review; integrated and installed at 6d2e0ba. Remaining UX and R4/R5 dependencies stay open."}, {"id": "R7", "title": "Complete Council within existing runner", "status": "pending", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"]}, diff --git a/docs/plans/engineering-team/evidence/R3-closure-2026-10-02.json b/docs/plans/engineering-team/evidence/R3-closure-2026-10-02.json new file mode 100644 index 0000000..4eae4b5 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R3-closure-2026-10-02.json @@ -0,0 +1,44 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "accepted_source_package", + "target_revision": "39b95f1c89ca89a73148cd88dfe5565f4d5dc377", + "independent_review": { + "run_id": "f0a4f86b-b305-4839-88a2-cb5fed8739ab", + "base_revision": "672e38305fcc764d87d7da6afb5914cbf3c5a3ce", + "state": "succeeded", + "version": 22, + "candidate_sha256": "010aa83f442fb6f275c1793b9131e5769b4481fd905ad9a1e83163ac1fc7c315", + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "verdict": "clean", + "findings": [], + "review_sha256": "c2a03b32bc382c2a5f9985d0a71e15e847b10fbe11bca93dc65f12b3ee413ccc", + "checks_sha256": "60567b4ef724ce176ed6bdba63164d182c5ce49e94524754edab07c08b16524f", + "receipt_sha256": "b013baa9fc4979c3dd979286753131067cc8e5d44f6f37b0f3e8d48a5cceb552", + "mandatory_checks": {"diff": "passed", "bash": "passed", "focused": {"tests": 51, "seconds": 200.684, "result": "pass"}, "candidate_integrity": "verified_unchanged_for_all_checks"}, + "usage": {"source": "native_reported", "input_tokens": 337192, "output_tokens": 2173, "total_tokens": 339365} + }, + "full_integration": { + "run_id": "360a4993-d016-406e-8bb4-ae6d57dfe6cf", + "candidate_revision": "f8c4f8c83f170870eb37d564e54eee6188fc233c", + "tests": 477, + "seconds": 455.860, + "skips": 2, + "errors": 0, + "failures": 0, + "unraisable": [], + "applicability": "git diff --exit-code confirms all six R3 source-module blobs identical between the accepted full-gate candidate and the independent audit target; no duplicate full run required" + }, + "requirements": [ + {"requirement": "immutable prelaunch and independent paired assignment", "proof": ["test_experiment_saved_runs", "test_experiment_evidence_integrity.test_trial_reservation_rejects_changed_inputs_before_any_worker", "test_experiment_v2_evaluation"], "result": "pass"}, + {"requirement": "branch-review and issue-delivery stream/input/import provenance", "proof": ["test_experiment_evidence_integrity", "test_delivery_experiment_integrity"], "result": "pass"}, + {"requirement": "honest failed, missing, fallback and cancelled arms", "proof": ["test_experiment_saved_runs", "test_delivery_experiment_integrity.test_real_failed_writer_requires_an_exact_terminal_receipt"], "result": "pass"}, + {"requirement": "unmeasured ratios cannot satisfy finite qualification gates", "proof": ["test_experiment_eligibility.test_unmeasured_ratios_cannot_bypass_finite_qualification_gates"], "result": "pass"}, + {"requirement": "correction after evaluation and qualification blocks new authority atomically", "proof": ["test_experiment_eligibility.test_correction_after_evaluation_blocks_new_qualification_atomically", "test_experiment_eligibility.test_correction_after_qualification_blocks_replay_and_new_promotion", "test_experiment_eligibility.test_correction_cannot_commit_between_qualification_validation_and_commit"], "result": "pass"}, + {"requirement": "valid promotion/rollback, stale rollback and catalog fallback gates", "proof": ["test_lifecycle.test_qualification_cas_promotion_and_rollback_are_replay_safe", "test_lifecycle.test_catalog_unavailable_incumbent_uses_only_qualified_predecessor", "test_experiment_eligibility.test_stale_regression_cannot_roll_back_until_explicit_current_review", "test_experiment_eligibility.test_stale_qualified_target_blocks_regression_and_is_skipped_by_catalog_fallback"], "result": "pass"}, + {"requirement": "append-only explicit revision, historical completed replay and duplicate-outcome upgrade read/proposal paths", "proof": ["test_experiment_eligibility.test_explicit_revision_keeps_original_bytes_and_reuses_original_assignments", "test_experiment_eligibility.test_completed_decision_replay_preserves_historical_receipt_after_correction", "test_experiment_eligibility.test_upgraded_reused_outcome_history_remains_readable_but_cannot_authorize_new_decisions"], "result": "pass"} + ], + "preserved_history": ["R3-audit-repairs-2026-10-01.json", "R3b2-eligibility-partial-2026-10-01.json", "R8-upgrade-review-2026-10-01.json", "R8-installed-workflows-2026-10-01.json"], + "separate_open_scope": ["R4 catalog/quota package acceptance", "R5 automatic final outcomes and public trial controller", "R6 terminal UX/readiness", "R7 Council", "R8 whole-delivery and desktop UI acceptance"], + "qualification_note": "Public offline fixtures establish machinery/provenance, not real-model superiority; original finite sample and budget gates remain unchanged" +} diff --git a/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json new file mode 100644 index 0000000..f70512e --- /dev/null +++ b/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json @@ -0,0 +1,32 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "baseline_revision": "39b95f1", + "status": "source_focused_verified_full_and_independent_gates_pending", + "scope": "Normal CLI native subscription discovery, private account/config/binary/version-scoped last-good catalog, 24-hour TTL, single OS-lock refresh owner, bounded pagination deadline, two-minute failure backoff, native typed quota ingestion and affected alias requalification block", + "verified": [ + {"gate": "native cache, normal entry, protocol and capacity", "tests": 79, "seconds": 14.696, "result": "pass"}, + {"gate": "two-project public normal review entry", "result": "pass", "scope": "Actual CLI preparation ingests quota and terminalizes both projects before any worker launch when the shared weekly window is exhausted despite available primary capacity"}, + {"gate": "real non-generating native discovery", "result": "pass", "scope": "Installed verified Codex binary, native account/config/model/quota read RPCs, exact gpt-6.1-sol/low, scoped pool and catalog fingerprint; two saved native windows yielded unknown capacity, not guessed availability"}, + {"gate": "privacy and isolation", "result": "pass", "scope": "Only hashes of account and effective configuration persist; incompatible account/config/binary/version scope cannot reuse catalogs. Failed quota query preserves still-fresh known exhaustion; a new incompatible account remains unknown"}, + {"gate": "provider default hint", "result": "pass", "scope": "Default-hint-only changes do not invalidate approved capability fingerprints or alter alias policy"} + ], + "intermediate_failures": [ + "Initial red test could not import the not-yet-implemented native_catalog module", + "Initial quota assertion lacked the required harness target selector; test corrected to query the actual Codex target", + "Initial cross-project fixture created Git repositories inside a blanket Popen mock; corrected selective native-process mock and created committed fixtures before mocking", + "Initial first-refresh backoff check accidentally caught its own ContractError (a ValueError subclass); separated the backoff decision from JSON validation", + "One overlapping focused invocation saw changing paired input bytes; final focused invocation used frozen source and passed. Full gate must always freeze source/tests" + ], + "constraints": { + "generating_calls": 0, + "login_or_config_mutations": 0, + "reset_credit_or_purchase_calls": 0, + "paid_api_fallback": false, + "manual_provider_observations_preserved": true, + "existing_aliases_mutated": false, + "explicit_pins_preserved": true + }, + "open_gates": ["R3c independent closure audit", "one full source integration gate", "bounded independent R4 review", "fresh installed normal-command recheck"], + "r3_review": {"run_id": "f0a4f86b-b305-4839-88a2-cb5fed8739ab", "target_revision": "39b95f1", "status": "running_at_checkpoint"} +} diff --git a/plugin/core/src/devsquad/catalog.py b/plugin/core/src/devsquad/catalog.py index f6977c0..bb5091d 100644 --- a/plugin/core/src/devsquad/catalog.py +++ b/plugin/core/src/devsquad/catalog.py @@ -68,7 +68,10 @@ def validate_catalog_change(value: dict[str, Any]) -> dict[str, Any]: def model_fingerprint(harness: str, version: str | None, model: dict[str, Any]) -> str: - stable = {"harness": harness, "version": version, "model": model} + # Provider ordering/default hints are not capability or serving revisions. + # Their movement must not invalidate an explicitly approved alias. + capabilities = {key: value for key, value in model.items() if key not in {"isDefault", "is_default"}} + stable = {"harness": harness, "version": version, "model": capabilities} return hashlib.sha256(json.dumps(stable, sort_keys=True, separators=(",", ":")).encode()).hexdigest() diff --git a/plugin/core/src/devsquad/native_catalog.py b/plugin/core/src/devsquad/native_catalog.py index e590afc..555196f 100644 --- a/plugin/core/src/devsquad/native_catalog.py +++ b/plugin/core/src/devsquad/native_catalog.py @@ -76,6 +76,7 @@ class NativeCatalogCache: def __init__(self, directory: Path, scope: str, version: str): self.directory, self.scope, self.version = directory, scope, version self.path = directory / (hashlib.sha256(scope.encode()).hexdigest() + ".json") + self.failure_path = self.path.with_suffix(".failure.json") def _read(self) -> dict[str, Any] | None: if not self.path.exists(): @@ -97,15 +98,16 @@ def _recent(timestamp: str, current: datetime, ttl: timedelta) -> bool: except (TypeError, ValueError): return False - def _write(self, value: dict[str, Any]) -> None: - temporary = self.path.with_suffix(f".tmp-{uuid.uuid4().hex}") + def _write(self, value: dict[str, Any], path: Path | None = None) -> None: + destination = path or self.path + temporary = destination.with_suffix(f".tmp-{uuid.uuid4().hex}") try: with temporary.open("x", encoding="utf-8") as output: os.chmod(temporary, 0o600) output.write(canonical_json(value) + "\n") output.flush() os.fsync(output.fileno()) - os.replace(temporary, self.path) + os.replace(temporary, destination) finally: temporary.unlink(missing_ok=True) @@ -123,6 +125,14 @@ def refresh(self, fetch: Callable[[], list[dict[str, Any]]], *, now: datetime | return previous try: previous = self._read() + if previous is None and self.failure_path.exists(): + try: + failure = json.loads(self.failure_path.read_bytes()) + backing_off = self._recent(failure["at"], current, REFRESH_BACKOFF) + except (ValueError, KeyError, TypeError): + raise ContractError("native discovery backoff evidence is invalid") from None + if backing_off: + raise ContractError("native discovery is backing off; retry shortly") if previous is not None and ( self._recent(previous["fetched_at"], current, CATALOG_TTL) or self._recent(previous["last_refresh"]["at"], current, REFRESH_BACKOFF) @@ -135,6 +145,7 @@ def refresh(self, fetch: Callable[[], list[dict[str, Any]]], *, now: datetime | models = normalize_models("codex", self.version, raw_models) except (ContractError, EOFError, OSError, TimeoutError): if previous is None: + self._write({"at": current.isoformat(), "status": "error"}, self.failure_path) raise ContractError("native discovery failed with no scoped last-good catalog") from None previous["last_refresh"] = {"at": current.isoformat(), "status": "error", "error": "discovery_failed"} self._write(previous) @@ -147,6 +158,7 @@ def refresh(self, fetch: Callable[[], list[dict[str, Any]]], *, now: datetime | } value["catalog_change"] = analyze_catalog_drift(previous, value) self._write(value) + self.failure_path.unlink(missing_ok=True) return value finally: fcntl.flock(lease, fcntl.LOCK_UN) diff --git a/plugin/core/src/devsquad/task_entry.py b/plugin/core/src/devsquad/task_entry.py index 19f6a83..686d985 100644 --- a/plugin/core/src/devsquad/task_entry.py +++ b/plugin/core/src/devsquad/task_entry.py @@ -151,9 +151,11 @@ def read_native(request_id, method, params=None): try: limits = read_native(1000, "account/rateLimits/read") except (ContractError, EOFError, OSError, TimeoutError): - limits = {} + # A failed query is not evidence that a still-fresh exhausted + # window disappeared. Retain prior observations until expiry. + limits = None service = Service(runtime) - for observation in normalize_codex_limits(limits, pool_id): + for observation in (normalize_codex_limits(limits, pool_id) if limits is not None else []): service.capacity_observe(observation) except (EOFError, OSError, TimeoutError) as exc: raise ContractError("Codex model discovery did not complete") from exc diff --git a/test/core/test_native_catalog.py b/test/core/test_native_catalog.py index 88f7ab8..a942e51 100644 --- a/test/core/test_native_catalog.py +++ b/test/core/test_native_catalog.py @@ -81,6 +81,14 @@ def test_scope_is_private_and_changes_for_account_config_binary_version(self): other = NativeCatalogCache(Path(self.temp.name), "scope-b", "v-test") with self.assertRaises(ContractError): other.refresh(lambda: (_ for _ in ()).throw(TimeoutError()), now=NOW) + with self.assertRaisesRegex(ContractError, "backing off"): + other.refresh(lambda: self.fail("initial failure backoff ignored"), now=NOW + timedelta(seconds=1)) + + def test_provider_default_hint_does_not_change_capability_fingerprint(self): + first = self.cache.refresh(lambda: [{**MODELS[0], "isDefault": True}], now=NOW) + changed = self.cache.refresh(lambda: [{**MODELS[0], "isDefault": False}], now=NOW + timedelta(days=2)) + self.assertEqual(first["models"][0]["fingerprint"], changed["models"][0]["fingerprint"]) + self.assertEqual(changed["catalog_change"]["changed_model_ids"], []) def test_weekly_limit_blocks_available_primary_and_null_is_unknown(self): payload = {"rateLimitsByLimitId": {"codex": { diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py index 95a6f80..04bf19c 100644 --- a/test/core/test_task_entry.py +++ b/test/core/test_task_entry.py @@ -775,16 +775,24 @@ def test_normal_native_discovery_caches_across_projects_and_ingests_shared_quota process.poll.return_value = None runtime = Path(self.temp.name) / "runtime" second_repo = self._init_repo() + (second_repo / "README.md").write_text("Second project\n") + self._commit(second_repo, "second project") + real_popen = subprocess.Popen + def native_popen(argv, *positional, **keywords): + return process if argv[:2] == ["/fixture/codex", "app-server"] else real_popen(argv, *positional, **keywords) reset = int((datetime.now(timezone.utc) + timedelta(days=2)).timestamp()) account = {"type": "chatgpt", "email": "fixture@example.invalid", "planType": "plus"} + failed_limits = False def reply(peer, request_id, **unused): + if request_id == 1000 and failed_limits: + raise TimeoutError("private provider quota diagnostic") return {"result": {1: {}, 2: {"account": account}, 3: {"config": {"provider": "native"}}, 1000: {"rateLimits": {"primary": {"usedPercent": 5, "windowDurationMins": 300, "resetsAt": reset}, "secondary": {"usedPercent": 100, "windowDurationMins": 10080, "resetsAt": reset}}}}[request_id]} with ( mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), - mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process), + mock.patch("devsquad.task_entry.subprocess.Popen", side_effect=native_popen), mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), mock.patch("devsquad.task_entry.receive_response", side_effect=reply), mock.patch("devsquad.task_entry.discover_models", return_value=[{"id": "gpt-fixture", "supportedReasoningEfforts": ["low"]}]) as discovery, @@ -793,6 +801,15 @@ def reply(peer, request_id, **unused): second = discover_codex_identity(second_repo, runtime=runtime) discovery.assert_called_once() self.assertEqual(first, second) + with mock.patch.object(Service, "_spawn_daemon") as spawn: + for project in (self.repo, second_repo): + output = io.StringIO() + with contextlib.redirect_stdout(output): + self.assertEqual(cli.main(["review", "--base", "HEAD", "--project-dir", str(project), "--runtime-dir", str(runtime), "--json"]), 0) + started = json.loads(output.getvalue())["data"] + self.assertEqual(started["state"], "failed") + self.assertEqual(started["service"]["error"]["error"], "CAPABILITY_UNAVAILABLE") + spawn.assert_not_called() service = Service(runtime) store = service._store() try: @@ -800,10 +817,22 @@ def reply(peer, request_id, **unused): self.assertEqual(store.capacity_snapshot(first["account_pool_id"], target=target)["status"], "exhausted") finally: store.close() + failed_limits = True + discover_codex_identity(self.repo, runtime=runtime) + store = service._store() + try: + self.assertEqual(store.capacity_snapshot(first["account_pool_id"], target=target)["status"], "exhausted") + finally: + store.close() account["email"] = "another@example.invalid" other = discover_codex_identity(self.repo, runtime=runtime) self.assertNotEqual(first["account_pool_id"], other["account_pool_id"]) self.assertEqual(discovery.call_count, 2) + store = service._store() + try: + self.assertEqual(store.capacity_snapshot(other["account_pool_id"], target=target)["status"], "unknown") + finally: + store.close() self.assertNotIn("fixture@example.invalid", "".join(path.read_text() for path in (runtime / "catalogs").glob("*.json"))) def test_normal_alias_rejects_account_or_catalog_change_without_mutating_binding(self): From 913abc1267e47684fee63dbbb1d69945f67f4a32 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 13:27:58 -0700 Subject: [PATCH 172/197] WIP checkpoint: R4 repair: one native subscription pool across catalog scopes (2026-10-02 13:27) --- docs/plans/engineering-team/RESUME.md | 17 ++++++++- docs/plans/engineering-team/backlog.json | 2 +- .../R4-native-scoped-partial-2026-10-02.json | 24 +++++++++++- plugin/core/src/devsquad/native_catalog.py | 17 +++++++-- plugin/core/src/devsquad/task_entry.py | 10 +++-- test/core/test_cli.py | 2 + test/core/test_native_catalog.py | 5 ++- test/core/test_task_entry.py | 38 ++++++++++++++++--- 8 files changed, 97 insertions(+), 18 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index a2627a4..559ed15 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -24,6 +24,19 @@ R3c independent saved review **`f0a4f86b-b305-4839-88a2-cb5fed8739ab`** is **succeeded/version 22**, accepted against immutable `39b95f1`; all mandatory checks and artifact hashes are verified. No active native audit remains. R4 partial evidence: `evidence/R4-native-scoped-partial-2026-10-02.json`. +R4 review **`9a12f6f6-59aa-43c0-a003-9f057f42f670`** was truthfully +rejected (failed/version 22): discovery config/binary/version scope must not +split one subscription-wide reservation fence. Repair now uses an opaque +account-only pool ID, while separate native-scope and catalog-fingerprint +evidence block incompatible approved aliases. Different-config cross-project +reservation and fresh-exhaustion regressions pass, plus CLI runtime-argument +expectations are updated. Initial full gate at `500d1b1` showed two CLI +expectation failures and was SIGINT-stopped (exit 130) to apply this finding; +it is not a completed gate. Its interrupted test setUp caused one implicit +TemporaryDirectory ResourceWarning. Latest 30-test targeted repair gate passes +in 2.712s; the combined **102-test** affected gate now passes in **17.610s**, +with explicit offline wheel-build interpreter and ResourceWarnings as errors. +Next: new frozen full integration gate and bounded pool/context repair review. Focused/full/independent package closure and fresh-install proof are still required before R4 is closed. One real non-generating native discovery probe verified gpt-6.1-sol/low, scoped pool/catalog fingerprint and two quota windows; @@ -130,8 +143,8 @@ Full portable redacted record: ## Broader plan / constraints M1/M2 remain accepted. R1/R2 source/offline repairs are verified; R3 saved-run -reader, eligibility/revisions, provenance/ratio and upgrade repairs have full -and bounded evidence, but final package-level closure audit remains open. +reader, eligibility/revisions, provenance/ratio and upgrade repairs are now +accepted with the independent R3 closure proof above. Normal aliases/public promotion proof exist; this does not close R4 catalog/ quota, R5 public trial controller/outcomes, remaining R6 UX or R7/C1 Council. Read backlog.json and SOL-REVIEW-FOLLOWUP.md for dependency/acceptance details; diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 3961d33..638b1d7 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -21,7 +21,7 @@ "status": "partial", "artifact": "evidence/R8-installed-workflows-2026-10-01.json", "scope": "Installed schema-15 runtime through 6d2e0ba; accepted real Claude-to-independent-Codex 477-test delivery, complete combined review and integration, actual Claude handoff/Grok operation and final Gemini CLI/MCP recheck", - "next_action": "Audit R3b.2/R3c closure against SOL-REVIEW-FOLLOWUP, then continue R4 catalog/quota, R5 public trials/outcomes, remaining R6 UX and R7/C1. Do not repeat accepted runtime proofs or unchanged full gates.", + "next_action": "R3 is accepted by evidence/R3-closure-2026-10-02.json. Finish R4's repaired account-wide pool fence full gate, narrow independent follow-up and installed recheck, then R5 public outcomes/trials, R6 UX and R7/C1. Preserve accepted earlier runtime proofs.", "limitations": "Whole-plan closure is not claimed. Required desktop UI proofs remain unverified; Antigravity IDE permission is denied. Grok/Gemini MCP status does not prove automatic writer/reviewer roles. Jev remains off with its one-request allowance spent; no paid API fallback or reset use authorized." }, "planning_checkpoint": { diff --git a/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json index f70512e..70cfa44 100644 --- a/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R4-native-scoped-partial-2026-10-02.json @@ -3,6 +3,28 @@ "recorded_on": "2026-10-02", "baseline_revision": "39b95f1", "status": "source_focused_verified_full_and_independent_gates_pending", + "review_finding_and_repair": { + "review_run_id": "9a12f6f6-59aa-43c0-a003-9f057f42f670", + "review_target": "500d1b1917ecbcf417af73fbfd21ae67accc45e7", + "state": "failed", + "version": 22, + "disposition": "reject", + "finding_id": "r4-account-pool-isolation", + "severity": "high", + "description": "Config/binary/version-scoped capacity pools partitioned a single subscription and bypassed max_concurrency=1", + "repair": "Use account-only opaque pool identity for shared observations/reservations, but keep separate account/config/binary/version native-scope and catalog-fingerprint qualification evidence", + "targeted_tests": {"tests": 30, "seconds": 2.712, "result": "pass"}, + "combined_affected_tests": {"tests": 102, "seconds": 17.610, "result": "pass", "installed_wheel_build_interpreter": "explicit_offline_python3.12", "resource_warnings": "errors"}, + "review_checks": {"focused_tests": 79, "seconds": 16.390, "result": "pass", "diff_and_bash": "pass", "integrity": "verified_unchanged"} + }, + "initial_full_gate": { + "target_revision": "500d1b1", + "result": "interrupted_not_passed", + "exit_code": 130, + "known_failures": "Two CLI mock call expectations omitted the newly required runtime argument; corrected without weakening preparation checks", + "reason": "Stopped exact full-test PID with SIGINT after known failures and independent high-severity finding, before source repair", + "unraisable_note": "Interruption during test setUp produced an implicit TemporaryDirectory ResourceWarning; no clean full-gate result is claimed" + }, "scope": "Normal CLI native subscription discovery, private account/config/binary/version-scoped last-good catalog, 24-hour TTL, single OS-lock refresh owner, bounded pagination deadline, two-minute failure backoff, native typed quota ingestion and affected alias requalification block", "verified": [ {"gate": "native cache, normal entry, protocol and capacity", "tests": 79, "seconds": 14.696, "result": "pass"}, @@ -28,5 +50,5 @@ "explicit_pins_preserved": true }, "open_gates": ["R3c independent closure audit", "one full source integration gate", "bounded independent R4 review", "fresh installed normal-command recheck"], - "r3_review": {"run_id": "f0a4f86b-b305-4839-88a2-cb5fed8739ab", "target_revision": "39b95f1", "status": "running_at_checkpoint"} + "r3_review": {"run_id": "f0a4f86b-b305-4839-88a2-cb5fed8739ab", "target_revision": "39b95f1", "status": "succeeded", "version": 22, "closure_evidence": "R3-closure-2026-10-02.json"} } diff --git a/plugin/core/src/devsquad/native_catalog.py b/plugin/core/src/devsquad/native_catalog.py index 555196f..109157b 100644 --- a/plugin/core/src/devsquad/native_catalog.py +++ b/plugin/core/src/devsquad/native_catalog.py @@ -24,17 +24,28 @@ QUOTA_TTL = timedelta(seconds=60) -def native_scope(account_result: dict[str, Any], config: dict[str, Any], binary: str, version: str) -> str: +def _account_identity(account_result: dict[str, Any]) -> dict[str, Any]: account = account_result.get("account") if not isinstance(account, dict) or account.get("type") != "chatgpt": raise ContractError("normal entry requires native ChatGPT subscription authentication") - identity = {key: account.get(key) for key in ("type", "id", "accountId", "email", "planType")} + identity = {key: account.get(key) for key in ("type", "id", "accountId", "email")} if not any(isinstance(identity[key], str) and identity[key] for key in ("id", "accountId", "email")): raise ContractError("native account identity is unknown; discovery cannot be reused") + return identity + + +def native_account_pool(account_result: dict[str, Any]) -> str: + """One reservation fence for one subscription, across discovery contexts.""" + return "codex-subscription-" + hashlib.sha256(canonical_json(_account_identity(account_result)).encode()).hexdigest() + + +def native_scope(account_result: dict[str, Any], config: dict[str, Any], binary: str, version: str) -> str: + identity = _account_identity(account_result) # Hash in memory only. Neither account identifiers nor effective configuration # (which can contain sensitive provider fields) are persisted or displayed. return hashlib.sha256(canonical_json({ - "account": identity, "config": config, "binary": binary, "version": version, + "account": identity, "plan_type": account_result["account"].get("planType"), + "config": config, "binary": binary, "version": version, }).encode()).hexdigest() diff --git a/plugin/core/src/devsquad/task_entry.py b/plugin/core/src/devsquad/task_entry.py index 686d985..43083c3 100644 --- a/plugin/core/src/devsquad/task_entry.py +++ b/plugin/core/src/devsquad/task_entry.py @@ -126,7 +126,7 @@ def discover_codex_identity( discover_models(peer, first_request_id=10, timeout_seconds=timeout_seconds), ) else: - from .native_catalog import NativeCatalogCache, native_scope, normalize_codex_limits + from .native_catalog import NativeCatalogCache, native_account_pool, native_scope, normalize_codex_limits from .service import Service deadline = time.monotonic() + timeout_seconds @@ -147,7 +147,7 @@ def read_native(request_id, method, params=None): lambda: discover_models(peer, first_request_id=10, timeout_seconds=max(0.001, deadline - time.monotonic())), ) models = catalog["models"] - pool_id = f"codex-subscription-{scope}" + pool_id = native_account_pool(account) try: limits = read_native(1000, "account/rateLimits/read") except (ContractError, EOFError, OSError, TimeoutError): @@ -193,7 +193,7 @@ def read_native(request_id, method, params=None): "model_id": selected["id"], "model_family": family if isinstance(family, str) and family else "gpt", "effort": effort, - **({"account_pool_id": pool_id, "catalog_fingerprint": selected["fingerprint"]} if scope is not None else {}), + **({"account_pool_id": pool_id, "native_scope": scope, "catalog_fingerprint": selected["fingerprint"]} if scope is not None else {}), } @@ -247,6 +247,7 @@ def _managed_routing( "evidence_refs": [ f"runtime-catalog:{codex['harness_version']}:{codex['model_id']}", *([f"runtime-catalog-fingerprint:{codex['catalog_fingerprint']}"] if "catalog_fingerprint" in codex else []), + *([f"runtime-native-scope:{codex['native_scope']}"] if "native_scope" in codex else []), ], } trial_profiles = {"reviewer": reviewer} @@ -291,6 +292,7 @@ def _managed_routing( if role == "reviewer" and "catalog_fingerprint" in codex and ( incumbent["account_pool_id"] != codex["account_pool_id"] or f"runtime-catalog-fingerprint:{codex['catalog_fingerprint']}" not in incumbent["evidence_refs"] + or ("native_scope" in codex and f"runtime-native-scope:{codex['native_scope']}" not in incumbent["evidence_refs"]) ): raise ContractError("approved reviewer requires requalification for the current native account/config/catalog") if incumbent["id"] == profile["id"] and incumbent != profile: @@ -506,7 +508,7 @@ def build_managed_task( "harness", "harness_version", "model_id", "model_family", "effort", } if (not isinstance(codex_identity, dict) - or set(codex_identity) - (required_identity | {"account_pool_id", "catalog_fingerprint"}) + or set(codex_identity) - (required_identity | {"account_pool_id", "native_scope", "catalog_fingerprint"}) or required_identity - set(codex_identity) or codex_identity.get("harness") != "codex" or not all( diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 8acb256..48b79c8 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -114,6 +114,7 @@ def test_review_dry_run_prepares_a_managed_task_without_starting(self): resolve.assert_called_once_with(str(self.root)) discover.assert_called_once_with( self.root, requested_model="gpt-fixture", requested_effort="low", + runtime=self.runtime, ) build.assert_called_once_with( workflow="branch-review", project_dir=self.root, @@ -175,6 +176,7 @@ def test_fix_starts_managed_delivery_and_reports_run_and_plan(self): service.start.assert_called_once_with(task, "fix-1", None) discover.assert_called_once_with( self.root, requested_model="gpt-review", requested_effort="high", + runtime=self.runtime, ) build.assert_called_once_with( workflow="issue-delivery", project_dir=self.root, diff --git a/test/core/test_native_catalog.py b/test/core/test_native_catalog.py index a942e51..e30e003 100644 --- a/test/core/test_native_catalog.py +++ b/test/core/test_native_catalog.py @@ -10,7 +10,7 @@ sys.path.insert(0, str(ROOT / "plugin/core/src")) from devsquad.contracts import ContractError -from devsquad.native_catalog import NativeCatalogCache, native_scope, normalize_codex_limits +from devsquad.native_catalog import NativeCatalogCache, native_account_pool, native_scope, normalize_codex_limits from devsquad.capacity import derive_pool_capacity @@ -67,6 +67,9 @@ def interrupted(): def test_scope_is_private_and_changes_for_account_config_binary_version(self): account = {"account": {"type": "chatgpt", "email": "private@example.invalid", "planType": "plus"}} scope = native_scope(account, {"provider": "native"}, "/binary/a", "v1") + pool = native_account_pool(account) + self.assertEqual(pool, native_account_pool({"account": {**account["account"], "planType": "pro"}})) + self.assertNotEqual(pool, native_account_pool({"account": {"type": "chatgpt", "email": "other@example.invalid"}})) for a, c, b, v in ( ({"account": {"type": "chatgpt", "email": "other@example.invalid"}}, {}, "/binary/a", "v1"), (account, {"provider": "changed"}, "/binary/a", "v1"), diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py index 04bf19c..6cd8392 100644 --- a/test/core/test_task_entry.py +++ b/test/core/test_task_entry.py @@ -22,7 +22,7 @@ parse_checks, ) from devsquad.router import load_routing -from devsquad.store import Store, canonical_json +from devsquad.store import ConflictError, Store, canonical_json from devsquad import cli @@ -782,13 +782,15 @@ def native_popen(argv, *positional, **keywords): return process if argv[:2] == ["/fixture/codex", "app-server"] else real_popen(argv, *positional, **keywords) reset = int((datetime.now(timezone.utc) + timedelta(days=2)).timestamp()) account = {"type": "chatgpt", "email": "fixture@example.invalid", "planType": "plus"} + configuration = {"config": {"provider": "native"}} + weekly_used = 100 failed_limits = False def reply(peer, request_id, **unused): if request_id == 1000 and failed_limits: raise TimeoutError("private provider quota diagnostic") - return {"result": {1: {}, 2: {"account": account}, 3: {"config": {"provider": "native"}}, + return {"result": {1: {}, 2: {"account": account}, 3: configuration, 1000: {"rateLimits": {"primary": {"usedPercent": 5, "windowDurationMins": 300, "resetsAt": reset}, - "secondary": {"usedPercent": 100, "windowDurationMins": 10080, "resetsAt": reset}}}}[request_id]} + "secondary": {"usedPercent": weekly_used, "windowDurationMins": 10080, "resetsAt": reset}}}}[request_id]} with ( mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), @@ -824,24 +826,48 @@ def reply(peer, request_id, **unused): self.assertEqual(store.capacity_snapshot(first["account_pool_id"], target=target)["status"], "exhausted") finally: store.close() + configuration["config"]["default_model"] = "changed" + configured = discover_codex_identity(second_repo, runtime=runtime) + self.assertEqual(configured["account_pool_id"], first["account_pool_id"]) + self.assertNotEqual(configured["native_scope"], first["native_scope"]) + store = service._store() + try: + self.assertEqual(store.capacity_snapshot(configured["account_pool_id"], target=target)["status"], "exhausted") + finally: + store.close() account["email"] = "another@example.invalid" other = discover_codex_identity(self.repo, runtime=runtime) self.assertNotEqual(first["account_pool_id"], other["account_pool_id"]) - self.assertEqual(discovery.call_count, 2) + self.assertEqual(discovery.call_count, 3) store = service._store() try: self.assertEqual(store.capacity_snapshot(other["account_pool_id"], target=target)["status"], "unknown") finally: store.close() + account["email"] = "fixture@example.invalid" + weekly_used, failed_limits = 5, False + available = discover_codex_identity(second_repo, runtime=runtime) + store = service._store() + try: + self.assertEqual(store.capacity_snapshot(available["account_pool_id"], target=target)["status"], "available") + one = store.claim_start(self.repo, "native-pool-fence-a", {"task": {}}, "fixture-a") + two = store.claim_start(second_repo, "native-pool-fence-b", {"task": {}}, "fixture-b") + store.reserve_pool_capacity(one.run_id, first["account_pool_id"], "qualification", target=target) + with self.assertRaises(ConflictError): + store.reserve_pool_capacity(two.run_id, configured["account_pool_id"], "qualification", target=target) + store.cancel_preparing(one.run_id) + store.cancel_preparing(two.run_id) + finally: + store.close() self.assertNotIn("fixture@example.invalid", "".join(path.read_text() for path in (runtime / "catalogs").glob("*.json"))) def test_normal_alias_rejects_account_or_catalog_change_without_mutating_binding(self): - identity = {**self.codex, "account_pool_id": "native-scope", "catalog_fingerprint": "a" * 64} + identity = {**self.codex, "account_pool_id": "native-pool", "native_scope": "scope-a", "catalog_fingerprint": "a" * 64} task, _ = build_managed_task(workflow="branch-review", project_dir=self.repo, base_ref="HEAD", target_ref="HEAD", goal="Bounded review", codex_identity=identity) incumbent = copy.deepcopy(task["routing"]["profiles"]["profiles"][0]) incumbent["quality_status"] = "proven" binding = {"alias": "review.deep", "version": 7, "profile": incumbent} - for change in ({"account_pool_id": "other-scope"}, {"catalog_fingerprint": "b" * 64}): + for change in ({"account_pool_id": "other-pool"}, {"native_scope": "scope-b"}, {"catalog_fingerprint": "b" * 64}): with self.assertRaisesRegex(ContractError, "requalification"): build_managed_task(workflow="branch-review", project_dir=self.repo, base_ref="HEAD", target_ref="HEAD", goal="Bounded review", codex_identity={**identity, **change}, role_bindings={"reviewer": binding}) self.assertEqual(binding["version"], 7) From d3f7c0b3cc1c704dee32f96a87500a3284572d9b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 13:41:26 -0700 Subject: [PATCH 173/197] WIP checkpoint: Accept R4 routing/catalog/quota and installed transport proofs (2026-10-02 13:41) --- docs/plans/engineering-team/RESUME.md | 84 ++++++++++--------- docs/plans/engineering-team/backlog.json | 2 +- .../evidence/R4-closure-2026-10-02.json | 46 ++++++++++ 3 files changed, 92 insertions(+), 40 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R4-closure-2026-10-02.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 559ed15..9724791 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -10,38 +10,44 @@ R3c is now **accepted** at immutable `39b95f1`: native independent audit is clean; mandatory diff/Bash/51 focused tests pass with unchanged integrity. The R3 source blobs exactly match the prior accepted 477-test full candidate. See `evidence/R3-closure-2026-10-02.json` for the requirement/proof matrix. -A coherent **partial R4** source slice is checkpointed: -native normal commands now call account/config/model/quota read-only RPCs; -scoped private last-good catalogs use a 24-hour TTL, OS refresh lease and -two-minute failure backoff. Native windows enter the shared typed capacity -ledger. Alias account/config/catalog mismatches block for requalification; -explicit pins remain trials. No login, reset-credit or API fallback calls. -New cache/concurrency/privacy/window and two-project discovery tests are added. -Latest stable-source focused gate: **79 tests passed in 14.696s**. Public CLI -entry for two committed projects proves fresh weekly exhaustion prevents any -worker launch; failed quota queries preserve still-fresh prior exhaustion. -R3c independent saved review **`f0a4f86b-b305-4839-88a2-cb5fed8739ab`** is -**succeeded/version 22**, accepted against immutable `39b95f1`; all mandatory -checks and artifact hashes are verified. No active native audit remains. -R4 partial evidence: `evidence/R4-native-scoped-partial-2026-10-02.json`. -R4 review **`9a12f6f6-59aa-43c0-a003-9f057f42f670`** was truthfully -rejected (failed/version 22): discovery config/binary/version scope must not -split one subscription-wide reservation fence. Repair now uses an opaque -account-only pool ID, while separate native-scope and catalog-fingerprint -evidence block incompatible approved aliases. Different-config cross-project -reservation and fresh-exhaustion regressions pass, plus CLI runtime-argument -expectations are updated. Initial full gate at `500d1b1` showed two CLI -expectation failures and was SIGINT-stopped (exit 130) to apply this finding; -it is not a completed gate. Its interrupted test setUp caused one implicit -TemporaryDirectory ResourceWarning. Latest 30-test targeted repair gate passes -in 2.712s; the combined **102-test** affected gate now passes in **17.610s**, -with explicit offline wheel-build interpreter and ResourceWarnings as errors. -Next: new frozen full integration gate and bounded pool/context repair review. -Focused/full/independent package closure and fresh-install proof are still -required before R4 is closed. One real non-generating native discovery probe -verified gpt-6.1-sol/low, scoped pool/catalog fingerprint and two quota windows; -reported capacity was unknown (not guessed available). Continue R4 full and -independent package/installed acceptance, then R5, R6 and R7/C1. +R4 is now **accepted and installed** at source `913abc1`: + +- Normal entry has private account/config/binary/version-scoped complete + last-good catalogs, 24h TTL, one OS refresh owner, bounded pagination and + two-minute failure backoff. Default hints do not promote or invalidate aliases. +- Native typed quota observations share one opaque **account-only** reservation + pool across discovery configurations. Incompatible qualified contexts block; + pins remain trials. Failed queries retain still-fresh known exhaustion. +- 102 affected tests in 17.610s; **484 full tests in 458.607s**, two optional SDK + skips, zero failures/errors/unraisable. Source was frozen for the full gate. +- Initial R4 audit `9a12f6f6…` rejected a high shared-pool partition bug. + Exact repair audit **`f45c723b-3067-473f-9349-f1012f5548e7`** is clean, + succeeded/22; diff/Bash/30 targeted tests pass, one optional wheel-environment + skip, all integrity hashes unchanged. Initial interrupted failed full gate + and the fixed CLI expectations remain truthfully recorded in partial evidence. +- Installed normal dry-run selects gpt-6.1-sol/low as a bounded trial, no run + creation; reinstall no-op, drift false, pip check pass, four registrations + matching. **Nine actual installed SDK transport tests pass in 2.743s, no skips.** + +Closure matrix: `evidence/R4-closure-2026-10-02.json`. No native audit, full +suite or nonterminal production run remains active. Earlier accepted native +Claude/Grok/Gemini proofs below remain historical; do not repeat unchanged ones. +Antigravity externally updated to **1.2.14**: current doctor correctly labels +its adapter unverified; the old 1.2.13 live status receipt is not a new-version +proof. Recheck this during R6/R8 without broad MCP listings or UI bypass. + +Next is **R5**: red public tests for objective idempotent final outcomes from +every terminal origin, including prelaunch failures/cancel, worker failures, +host/headless completion and crash/replay. Preserve missingness and failed +attempt/rework credit. Use a durable projection outbox with guarded migration +if needed; do not rewrite historical final outcomes or retroactively assign +legacy runs. Then expose a thin public trial controller reusing schema-14 +assignment authority and R3 eligibility. Replace the fixture's private +preparation monkeypatch/manual outcome import with that actual public path. +Freeze full spec/cases/splits/profiles/budgets before either arm, enforce budget +from durable attempts under the existing reservation fence, and keep automatic +experimentation disabled. Continue R6 UX/readiness/generated-reference checks, +R7/C1 and final R8 acceptance; whole-plan completion is not claimed. Workspace: /Users/Dikshant/Desktop/Projects/devsquad. Branch: `codex/engineering-team`; never restart this build from main. @@ -78,10 +84,10 @@ integrity; 27 affected tests in 24.173s. Verified native Codex review is clean. 1. The user's three requested runtime actions are verified. Do not repeat these accepted proofs or the unchanged full suite. Both exact packets and artifact hashes are verified; never edit frozen evidence. -2. Audit R3b.2/R3c closure against SOL-REVIEW-FOLLOWUP.md and the requirement - matrix, preserving existing reader/eligibility/compatibility repairs. Then - continue R4 catalog/quota, R5 public trials/outcomes, remaining R6 UX and - R7/C1 in dependency order. Whole-plan acceptance is not claimed. +2. R3 and R4 are accepted by their October 2 closure matrices. Continue R5 + public terminal outcomes/trial controller, then remaining R6 UX/readiness, + R7/C1 and R8 whole-delivery acceptance. Preserve the existing reader and + lifecycle eligibility gates; do not restart the architecture exercise. 3. Desktop UI proofs remain separate. Antigravity IDE control is permission- denied; do not bypass it or substitute a CLI receipt. Grok/Gemini MCP status calls do not prove automatic writer/reviewer adapters. @@ -94,7 +100,7 @@ use git-safety checkpoints, never stash. No goal is currently active. ## Current local installation and host proof Stable launcher: /Users/Dikshant/.local/bin/squad. -Selected release: `0.1.0-py31214-01fad439adea-mcp-a26bc88afbef`. +Selected release: `0.1.0-py31214-3631f1737bc9-mcp-a26bc88afbef`. Python 3.12.14/MCP 2.2.0, schema 15; previous releases and private pre-upgrade SQLite backup retained. Source/plugin/installed payload drift is false. Upgrade defers for old active/recoverable runs, swaps without migration under @@ -145,8 +151,8 @@ Full portable redacted record: M1/M2 remain accepted. R1/R2 source/offline repairs are verified; R3 saved-run reader, eligibility/revisions, provenance/ratio and upgrade repairs are now accepted with the independent R3 closure proof above. -Normal aliases/public promotion proof exist; this does not close R4 catalog/ -quota, R5 public trial controller/outcomes, remaining R6 UX or R7/C1 Council. +Normal aliases/catalog/quota are accepted by the R4 closure above. This does +not close R5 public trials/outcomes, remaining R6 UX or R7/C1 Council. Read backlog.json and SOL-REVIEW-FOLLOWUP.md for dependency/acceptance details; do not restart the architecture exercise or weaken gates to mark these done. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 638b1d7..764815e 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -45,7 +45,7 @@ {"id": "R1", "title": "Candidate integrity through trusted checks", "status": "complete", "milestones": ["M3", "M5"], "depends_on": [], "items": ["F1"], "evidence": "evidence/R1-candidate-integrity-2026-09-29.json", "checkpoint": "Source repair verified by public regressions, mutation matrix, 330-test core gate with 2 optional-SDK skips and independent patch review. Installed refresh remains R8."}, {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, - {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "in_progress", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-native-scoped-partial-2026-10-02.json", "checkpoint": "Partial scoped native cache/quota production connection passes 79 focused tests and a non-generating native probe. Two-project public CLI quota fence verified. R3 independent closure, full integration, independent R4 audit and installed recheck still required; not accepted."}, + {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"G4 discovery verified by accepted native 477-test delivery and combined review; integrated and installed at 6d2e0ba. Remaining UX and R4/R5 dependencies stay open."}, {"id": "R7", "title": "Complete Council within existing runner", "status": "pending", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"]}, diff --git a/docs/plans/engineering-team/evidence/R4-closure-2026-10-02.json b/docs/plans/engineering-team/evidence/R4-closure-2026-10-02.json new file mode 100644 index 0000000..722b83b --- /dev/null +++ b/docs/plans/engineering-team/evidence/R4-closure-2026-10-02.json @@ -0,0 +1,46 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "accepted_source_and_installed_routing_package", + "source_revision": "913abc1267e47684fee63dbbb1d69945f67f4a32", + "full_gate": {"tests": 484, "seconds": 458.607, "skips": 2, "errors": 0, "failures": 0, "unraisable": [], "monotonic_seconds": 458.6838033749991, "utc_seconds": 458.680991, "offline_build_interpreter": "explicit_python3.12", "source_and_tests_frozen": true}, + "focused_gate": {"tests": 102, "seconds": 17.610, "result": "pass"}, + "independent_repair_review": { + "run_id": "f45c723b-3067-473f-9349-f1012f5548e7", + "base_revision": "500d1b1917ecbcf417af73fbfd21ae67accc45e7", + "candidate_sha256": "15ca9655c33f89b9be71fcad07c82c5c996c5c77b7c7ea1c67f18141ccd61076", + "state": "succeeded", "version": 22, "disposition": "accept", "verdict": "clean", "findings": [], + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "review_sha256": "a3e3d56f5cc91b56737952eb53912d818dd914d4d7a08abf69ce6660ba9feba7", + "checks_sha256": "1bca1c29edd2337634a9a14d0d1d701ad1b6a6abbbbd2102bf95a13c6651160c", + "receipt_sha256": "7f3a50bf9c5a23064278b78cda4f8773ab40ab153d238ca88294778f188a5c76", + "checks": {"diff_and_bash": "pass", "focused_tests": 30, "seconds": 0.993, "optional_installed_wheel_environment_skips": 1, "integrity": "verified_unchanged"}, + "usage": {"source": "native_reported", "input_tokens": 95191, "output_tokens": 567, "total_tokens": 95758} + }, + "installation": { + "release": "0.1.0-py31214-3631f1737bc9-mcp-a26bc88afbef", + "source_digest": "3631f1737bc97132dcb51596c65b929c13b1a0904f2f9776a586f106e6c53b9d", + "python": "3.12.14", "mcp": "2.2.0", "ledger_schema": 15, + "nonterminal_runs_before": 0, "first_changed": true, "reinstall_changed": false, + "network_downloads": false, "all_payload_drift": false, "manifest_matches": true, + "pip_check": "pass", "previous_releases_retained": true, + "installed_sdk_transport": {"tests": 9, "seconds": 2.743, "skips": 0, "result": "pass", "import_origin": "resolved installed release, not source"}, + "normal_cli_discovery": {"workflow": "branch-review", "dry_run": true, "model": "gpt-6.1-sol", "effort": "low", "selection_mode": "bounded_trial", "run_created": false, "task_sha256": "d026a58b367879fb61d754a0fd252e798a10a5e091acd71346bb206f9833a4e2"}, + "host_registrations": {"matching": 4, "writes": 0}, + "doctor_note": "Current doctor confirms installation/registration only, not authenticated workflow readiness; that distinction is R6. Antigravity externally changed to 1.2.14 and remains unverified; prior 1.2.13 live receipt is historical" + }, + "requirements": [ + {"requirement": "approved aliases survive provider-default hints; reviewed promotion changes only new runs; exact pins and frozen runs preserved", "proof": "test_task_entry public promotion and pin fixtures plus test_native_catalog default-hint fingerprint test", "result": "pass"}, + {"requirement": "scoped complete last-good catalog, TTL, one refresh owner, interruption, initial and prior-snapshot failure backoff", "proof": "test_native_catalog and production normal CLI discovery connection", "result": "pass"}, + {"requirement": "affected model/account/config/version changes block normal approved roles for requalification, without alias mutation or promotion from discovery", "proof": "test_task_entry native context/catalog mismatch and scoped discovery tests", "result": "pass"}, + {"requirement": "two projects honor weekly exhaustion despite available primary quota before worker launch", "proof": "test_task_entry actual normal CLI preparation over two committed projects; daemon launch asserted absent", "result": "pass"}, + {"requirement": "same account across discovery configurations shares one reservation fence and retains fresh exhausted quota", "proof": "test_task_entry cross-config public discovery and Store reservation fence regression; independent audit follow-up", "result": "pass"}, + {"requirement": "incompatible accounts do not reuse capacity; unsupported/null/stale observations remain unknown; manual observations retained", "proof": "test_native_catalog, test_capacity and cross-account normal discovery fixture", "result": "pass"}, + {"requirement": "no billing, permission, login, reset-credit or paid-API expansion", "proof": "read-only RPC allowlist and fixed normal role permissions; all discovery calls are non-generating", "result": "pass"} + ], + "official_native_protocol_reference": "https://learn.chatgpt.com/docs/app-server", + "preserved_failed_and_partial_history": "R4-native-scoped-partial-2026-10-02.json", + "operator_probe_note": "First installed-origin assertion compared a symlink path string instead of resolved path; corrected resolved-origin assertion passed the actual installed SDK gate. Expected negative squad_events argument rejection is not a failed test", + "separate_open_scope": ["R5 automatic terminal outcomes/public trials", "R6 terminal UX/auth readiness/generated-reference check", "R7 Council", "R8 final delivery/desktop UI/new Antigravity version acceptance"], + "adoption_note": "No real-model superiority or automatic policy promotion is claimed. Normal unqualified selection remains a bounded trial; Jev stays off" +} From 4d9e298308ea82410ae8e80e574cc36fb7aa5fde Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 14:07:23 -0700 Subject: [PATCH 174/197] WIP checkpoint: R5 WIP: objective terminal projections and public budget-fenced trial chain (2026-10-02 14:07) --- docs/RUNTIME-GUIDE.md | 25 +++ docs/generated/core-reference.md | 1 + docs/plans/engineering-team/CONTRACTS.md | 25 +++ docs/plans/engineering-team/RESUME.md | 35 ++-- docs/plans/engineering-team/backlog.json | 4 +- ...public-integration-partial-2026-10-02.json | 31 +++ plugin/core/src/devsquad/cli.py | 21 ++ .../migrations/016_objective_outcome_jobs.sql | 8 + .../core/src/devsquad/objective_outcomes.py | 101 ++++++++++ plugin/core/src/devsquad/service.py | 42 +++- plugin/core/src/devsquad/store.py | 98 +++++++++- test/core/experiment_runtime_fixture.py | 27 +-- test/core/test_cli.py | 23 ++- test/core/test_experiment_eligibility.py | 4 +- test/core/test_install_core.py | 7 +- test/core/test_lifecycle.py | 2 +- test/core/test_objective_outcomes.py | 143 ++++++++++++++ test/core/test_public_trials.py | 184 ++++++++++++++++++ test/core/test_service.py | 12 +- test/core/test_store.py | 10 +- 20 files changed, 744 insertions(+), 59 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json create mode 100644 plugin/core/src/devsquad/migrations/016_objective_outcome_jobs.sql create mode 100644 plugin/core/src/devsquad/objective_outcomes.py create mode 100644 test/core/test_objective_outcomes.py create mode 100644 test/core/test_public_trials.py diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index 4bd28e7..477f302 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -98,6 +98,31 @@ The lower-level automation path remains available. A hand-written task must name an existing Git repository and committed refs. It may use either committed routing files or an exact embedded registry/policy pair. +New public runs automatically record one final learning outcome, including +failed attempts and repairs. `squad report --project "$PWD" --json` includes +these without manual imports. Later feedback remains an explicit late +correction via `squad outcome add RUN --file correction.json --json`; it never +overwrites the original final outcome. Historical missing outcomes stay +missing rather than being manufactured during an upgrade. + +For an explicitly reviewed, predeclared comparison, start each bounded arm: + +```bash +squad trial --experiment experiment.json --case CASE --arm control \ + --task-file control-task.json --idempotency-key trial-CASE-control --wait --json +squad trial --experiment experiment.json --case CASE --arm candidate \ + --task-file candidate-task.json --idempotency-key trial-CASE-candidate --wait --json +``` + +This advanced automation command requires the v2 declaration (case/split, +input and concrete execution hashes, gates, budgets) before either arm. Use +branch reviews for reviewer comparisons or issue delivery for implementer +comparisons. All arms share the declared reservation/wall budget; failed and +fallback slots count. Automatic experimentation and promotion remain off. +The same `policy evaluate`, profile qualification and reviewed binding-change +commands operate on the resulting saved-run evidence; missing arms cannot +create a completed pair or authorize promotion. + ```bash squad start --task-file task.json --idempotency-key issue-123 --json squad status RUN_ID --json diff --git a/docs/generated/core-reference.md b/docs/generated/core-reference.md index 3fc28bc..d0c8ee3 100644 --- a/docs/generated/core-reference.md +++ b/docs/generated/core-reference.md @@ -39,6 +39,7 @@ Regenerate with `python3 scripts/generate-core-reference.py`; verify with - `squad setup [-h] [--host {codex,claude-code,antigravity,grok}] [--dry-run] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` - `squad start [-h] --task-file TASK_FILE --idempotency-key IDEMPOTENCY_KEY [--supersedes-run SUPERSEDES_RUN] [--wait] [--json] [--runtime-dir RUNTIME_DIR]` - `squad status [-h] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad trial [-h] --experiment EXPERIMENT --case CASE --arm {control,candidate} --task-file TASK_FILE --idempotency-key IDEMPOTENCY_KEY [--wait] [--json] [--runtime-dir RUNTIME_DIR]` ## Common command examples diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index e0bdd8f..1f80a08 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -268,6 +268,31 @@ Use this loop: Default experiment budget is disabled until explicitly configured, then at most 10% of eligible runs with a hard call/time cap. Most work uses the current proven policy. V1 uses human-governed static preferences, not exhaustive permutations, an automatic bandit or foundation-model fine-tuning. Report sample sizes and missingness; tiny samples justify hypotheses, not provider rankings. +New public runs (schema 16) request an objective final-outcome projection at +admission. Terminal receipts remain immutable. A durable outbox and exact +outcome ID make projection replay-safe after a crash; status/result, project +reporting and experiment evaluation repair only their relevant pending jobs. +Legacy history is not retrospectively assigned or rewritten. Prelaunch +failures/cancellations have no worker contributions. Completed worker failures +remain failed; subsequent fallback/revision success is repair, not independently +successful original work. Reviewer findings and explicit lead revisions remain +visible. Successful criterion status cites the fenced lead's final acceptance, +not a fabricated check result; unevaluated criteria stay unknown. Subjective +later feedback is an explicit append-only late correction, not another final. + +The opt-in `trial --experiment FILE --case ID --arm control|candidate +--task-file FILE --idempotency-key KEY` command starts one arm through the +ordinary runner. It freezes the complete v2 declaration under the existing +preparation fence before any attempt. Reviewer trials require branch-review's +same frozen candidate; implementer trials require issue-delivery's same +baseline/task/check contract. The shared reservation transaction counts all +durable experiment attempt slots (including failed, fallback, revision and +in-flight slots) against the declared call cap; the wall deadline starts at +declaration freeze. Controller limits are 100 cases, 1,000 reservations and +3,600 seconds, never an entitlement to spend that much. Missing/unstarted arms +do not become completed pairs. This explicit manual operation does not enable +automatic experimentation, dispatch a background campaign or promote a binding. + Measure acceptance and critical defects first; also show retries, lead rework, elapsed time, measured usage by pool, blocked time and unmeasured overhead. Final task success and original worker quality are distinct. Pair deterministic checks with review and human correction; a model judging itself is not sufficient evidence. ### Independent experiment provenance (v2) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 9724791..667a73f 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -36,18 +36,29 @@ Antigravity externally updated to **1.2.14**: current doctor correctly labels its adapter unverified; the old 1.2.13 live status receipt is not a new-version proof. Recheck this during R6/R8 without broad MCP listings or UI bypass. -Next is **R5**: red public tests for objective idempotent final outcomes from -every terminal origin, including prelaunch failures/cancel, worker failures, -host/headless completion and crash/replay. Preserve missingness and failed -attempt/rework credit. Use a durable projection outbox with guarded migration -if needed; do not rewrite historical final outcomes or retroactively assign -legacy runs. Then expose a thin public trial controller reusing schema-14 -assignment authority and R3 eligibility. Replace the fixture's private -preparation monkeypatch/manual outcome import with that actual public path. -Freeze full spec/cases/splits/profiles/budgets before either arm, enforce budget -from durable attempts under the existing reservation fence, and keep automatic -experimentation disabled. Continue R6 UX/readiness/generated-reference checks, -R7/C1 and final R8 acceptance; whole-plan completion is not claimed. +**R5 is source-in-progress**, not accepted/installed. The new schema-16 +projection outbox opts in new public runs without rewriting legacy history. +Terminal transitions and targeted status/result/report/evaluation replay +generate one final outcome; failures, repairs, findings, fenced lead +attestations and late corrections remain separate. Corrupt pending evidence +does not block unrelated runs. The explicit public `trial` / `trial_start` +path now uses the schema-14 assignment fence; the fixture no longer patches +preparation or manually imports finals. Experiment reservations share a +durable all-attempt call cap and a declaration-time wall deadline; automatic +experimentation stays off. 88 affected outcome/service/CLI/store tests passed +in 61.236s (before the added deadline test/lead-repair attribution refinement). +Public concurrent-budget and promotion → new-run binding → held-out regression +→ rollback tests pass. See `evidence/R5-public-integration-partial-2026-10-02.json`. + +Next: finish the frozen learning/lifecycle affected gate currently running +(exec session 99417; do not start a second copy or edit source/tests during +it). Then full core gate with explicit offline Python 3.12 build interpreter, +independent exact-patch review, schema-16 safe installer/installed transport +recheck and R5 closure. Current production installation/ledger is still the +accepted R4 **schema 15**; never open it with source Service while schema-16 +work is incomplete. Preserve failed red/intermediate probes truthfully. +Continue R6 UX/readiness/generated-reference checks, R7/C1 and final R8 +acceptance after R5; whole-plan completion is not claimed. Workspace: /Users/Dikshant/Desktop/Projects/devsquad. Branch: `codex/engineering-team`; never restart this build from main. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 764815e..45b8cae 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -14,7 +14,7 @@ "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", "next_milestone": "M5", - "next_work_package": "R3b.2", + "next_work_package": "R5", "partial_implementation_checkpoint": { "recorded_on": "2026-10-01", "baseline_revision": "6d2e0ba", @@ -46,7 +46,7 @@ {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, - {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "pending", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"]}, + {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "in_progress", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-public-integration-partial-2026-10-02.json", "checkpoint": "Schema-16 objective projection and explicit public trial/controller connections implemented, not yet accepted or installed. 88 affected tests pass; public concurrent-budget and promotion/new-run/held-out-regression/rollback chain pass. Frozen learning/lifecycle gate running; full integration, independent review and installed upgrade/transport proof remain open. No automatic experimentation or provider-quality claim."}, {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"G4 discovery verified by accepted native 477-test delivery and combined review; integrated and installed at 6d2e0ba. Remaining UX and R4/R5 dependencies stay open."}, {"id": "R7", "title": "Complete Council within existing runner", "status": "pending", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"]}, {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} diff --git a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json new file mode 100644 index 0000000..58ccbcf --- /dev/null +++ b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json @@ -0,0 +1,31 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "source_partial_not_accepted_or_installed", + "base_revision": "d3f7c0b", + "source_schema": 16, + "installed_schema": 15, + "requirements": [ + {"requirement": "idempotent objective final projection from preparation failure/cancel, worker failure, host/headless completion and crash replay", "proof": "test_objective_outcomes public Service transitions", "result": "focused_pass"}, + {"requirement": "append-only late corrections, no manual public replacement final, scoped corruption isolation and public report replay", "proof": "test_objective_outcomes and test_service", "result": "focused_pass"}, + {"requirement": "public predeclared assignments and automatic finals, not preparation monkeypatch or manual import", "proof": "ExperimentRuntimeFixture now calls Service.trial_start; public reviewer and implementer pair tests", "result": "focused_pass"}, + {"requirement": "one shared experiment reservation cap across concurrent arms and fallback", "proof": "test_public_trials actual concurrent public resumes and fallback failure", "result": "focused_pass"}, + {"requirement": "evaluate, qualify, reviewed promotion for a new run, held-out regression and rollback", "proof": "test_public_trials complete Service API chain with actual offline worker processes", "result": "focused_pass"}, + {"requirement": "declaration-time experiment wall deadline and precise headless lead-repair attribution", "proof": "added regression/refinement after initial 88-test gate", "result": "affected_gate_running"} + ], + "passed_gates": [ + {"command": "python3 -m unittest test_objective_outcomes test_public_trials test_cli test_store test_service", "tests": 88, "seconds": 61.236, "resource_warnings": "promoted_to_errors", "offline_build_interpreter": "explicit_python3.12", "result": "pass_before_final_deadline_and_lead_attribution_refinement"}, + {"command": "python3 -m unittest test_public_trials.PublicTrialTest.test_public_outcomes_evaluate_qualify_promote_new_run_and_roll_back", "tests": 1, "seconds": 12.300, "result": "pass"} + ], + "failure_history": [ + "Initial seven public terminal-origin tests failed because no automatic finals existed (7 failures/9.322s).", + "First projection implementation had wrong criterion-status and incomplete worker-failure mapping (1 failure/4 errors in 7 tests/9.356s).", + "Headless receipt attempts do not include IDs on successful semantic records; keyed failure lookup was corrected (1 error in 7 tests/9.472s).", + "Existing generic fake-runner native exit receipt lacks managed run/state fields; explicitly fixture-gated normalization now preserves that original receipt (1 error in 3 tests/15.093s).", + "New whole-chain test used an invalid invented template mode human_reviewed; corrected to the existing reviewed contract, not a source policy relaxation (1 error in 6 tests/6.586s).", + "An exploratory fixture diagnostic requested a nonexistent exit_code column and failed before printing; the corrected diagnostic used finally cleanup. This is not a passing test or a production fault." + ], + "open_gates": ["learning/lifecycle affected suite", "full core suite", "independent exact-patch review", "schema-16 safe upgrade and installed transport proof", "R6/R7/R8 separate acceptance"], + "policy": {"automatic_experiments": "off", "automatic_promotion": "not_enabled", "new_provider_calls_for_this_slice": 0, "paid_api_fallback": false, "legacy_outcomes_rewritten": false}, + "limits": "Fixture outcomes demonstrate runtime/provenance behavior, not real model superiority or subscription cost savings. Installed runtime remains the accepted R4 release until acceptance gates pass." +} diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index a9a7efd..bfee1e4 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -183,6 +183,17 @@ def command_start(args: argparse.Namespace) -> tuple[dict, int]: return _wait_for_run(service, started) +def command_trial(args: argparse.Namespace) -> tuple[dict, int]: + service = _service(args) + started = service.trial_start( + _read_json(args.experiment, "experiment file"), args.case, args.arm, + _read_json(args.task_file, "trial task file"), args.idempotency_key, + ) + if args.wait: + return _wait_for_run(service, started, resume_candidate_review=True) + return envelope(data=started), 0 + + def _normal_entry_result( summary: dict[str, Any], idempotency_key: str, @@ -490,6 +501,16 @@ def parser() -> argparse.ArgumentParser: fix.add_argument("--runtime-dir", default=runtime_default) fix.set_defaults(func=command_fix) start = sub.add_parser("start"); start.add_argument("--task-file", required=True); start.add_argument("--idempotency-key", required=True); start.add_argument("--supersedes-run"); start.add_argument("--wait", action="store_true"); start.add_argument("--json", action="store_true"); start.add_argument("--runtime-dir", default=runtime_default); start.set_defaults(func=command_start) + trial = sub.add_parser("trial", help="explicitly run one predeclared bounded experiment arm") + trial.add_argument("--experiment", required=True) + trial.add_argument("--case", required=True) + trial.add_argument("--arm", choices=("control", "candidate"), required=True) + trial.add_argument("--task-file", required=True) + trial.add_argument("--idempotency-key", required=True) + trial.add_argument("--wait", action="store_true") + trial.add_argument("--json", action="store_true") + trial.add_argument("--runtime-dir", default=runtime_default) + trial.set_defaults(func=command_trial) for name, fn in (("status",command_status),("result",command_result),("cancel",command_cancel),("resume",command_resume)): cmd=sub.add_parser(name); cmd.add_argument("run"); cmd.add_argument("--json",action="store_true"); cmd.add_argument("--runtime-dir",default=runtime_default) if name == "resume": cmd.add_argument("--recovery-file") diff --git a/plugin/core/src/devsquad/migrations/016_objective_outcome_jobs.sql b/plugin/core/src/devsquad/migrations/016_objective_outcome_jobs.sql new file mode 100644 index 0000000..7cfcc07 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/016_objective_outcome_jobs.sql @@ -0,0 +1,8 @@ +-- New public runs request objective projection at admission. Legacy final +-- outcomes and missing legacy history are not rewritten or retroactively armed. +CREATE TABLE objective_outcome_jobs ( + run_id TEXT PRIMARY KEY REFERENCES runs(id), + requested_at TEXT NOT NULL, + completed_outcome_id TEXT REFERENCES outcomes(outcome_id) +); +CREATE INDEX pending_objective_outcomes ON objective_outcome_jobs(completed_outcome_id, run_id); diff --git a/plugin/core/src/devsquad/objective_outcomes.py b/plugin/core/src/devsquad/objective_outcomes.py new file mode 100644 index 0000000..a419fa3 --- /dev/null +++ b/plugin/core/src/devsquad/objective_outcomes.py @@ -0,0 +1,101 @@ +"""Objective projection only; subjective later corrections stay append-only.""" +from __future__ import annotations + +import json +from datetime import datetime, timezone +from typing import Any + +from .contracts import ContractError + + +def project_outcome(run: dict[str, Any], receipt: dict[str, Any], attempts: list[dict[str, Any]], + artifact_refs: dict[str, str], assignment: dict[str, Any] | None) -> dict[str, Any]: + snapshot = json.loads(run["mutable_snapshot"] or "null") + if ("internal_fake_delay" in (snapshot or {}) and set(receipt) == { + "returncode", "cancelled", "timed_out", "stdout", "stderr", "finished_at"}): + # The existing generic runner fixture retains its native exit receipt, + # unlike the managed workflow's semantic receipt. Do not rewrite it. + if (type(receipt["returncode"]) is not int or type(receipt["cancelled"]) is not bool + or type(receipt["timed_out"]) is not bool + or type(receipt["finished_at"]) not in {int, float}): + raise ContractError("objective generic exit receipt is invalid") + expected = "cancelled" if receipt["cancelled"] else "failed" if receipt["timed_out"] or receipt["returncode"] != 0 else "succeeded" + if run["state"] != expected and run["state"] != "cancelled": + raise ContractError("objective generic exit contradicts terminal state") + receipt = {**receipt, "run_id": run["id"], "state": run["state"], + "finished_at": datetime.fromtimestamp(receipt["finished_at"], timezone.utc).isoformat(), + "attempt_id": attempts[-1]["id"] if attempts else None, + "error": {"returncode": receipt["returncode"]} if expected == "failed" else None} + if receipt.get("run_id") != run["id"] or receipt.get("state") != run["state"]: + raise ContractError("objective receipt does not match the terminal run") + roles = (snapshot or {}).get("routing", {}).get("roles", {}) + mode = ("experimental" if assignment is not None or (snapshot or {}).get("experiment_assignment") is not None else "pinned" + if any(role.get("source") == "override" for role in roles.values()) else "automatic") + terminal_ref = artifact_refs.get("receipt.json", artifact_refs["result-receipt.json"]) + accepted = run["state"] == "succeeded" and receipt.get("lead", {}).get("disposition") == "accept" + criteria = [] + for criterion in receipt.get("criteria", []): + # The receipt's evaluation supplies evidence, not a lead verdict. A + # fenced final acceptance attests the criteria; cite that attestation + # rather than manufacturing a check-level result from evidence_available. + criteria.append({ + "criterion_id": criterion["id"], + "status": "passed" if accepted else "unknown", + "evidence_refs": [terminal_ref] if accepted else [], + }) + reported_attempts = {item["id"]: item for item in + receipt.get("attempts", []) + receipt.get("lead", {}).get("attempts", []) + if item.get("id") is not None} + failed_roles, contributions = set(), [] + for attempt in attempts: + # Reservation and prelaunch cancellation are not worker exposure. + if attempt["status"] != "finished" or attempt.get("pid") is None: + continue + metadata = json.loads(attempt["output_metadata"] or "null") + reported = reported_attempts.get(attempt["id"], {}) + failed = (isinstance(metadata, dict) and metadata.get("failure") is not None + or reported.get("status") == "failed" or reported.get("error") is not None + or receipt.get("attempt_id") == attempt["id"] and receipt.get("error") is not None) + role = attempt["role"] + repaired = role in failed_roles + if failed: + result = "failed" + failed_roles.add(role) + elif repaired: + result = "repair" + elif role == "reviewer" and reported.get("review", {}).get("verdict") == "findings": + result = "finding" + elif run["state"] == "succeeded" and criteria and all(c["status"] == "passed" for c in criteria): + result = "successful" + else: + result = "neutral" + refs = [value for name, value in artifact_refs.items() if attempt["id"] in name] + [terminal_ref] + contributions.append({"attempt_id": attempt["id"], "role": role, "result": result, + "independent_success": result == "successful", "evidence_refs": refs}) + imported_leads = [a["id"] for a in attempts if a["role"] == "lead" + and f"lead-attempt-{a['id']}.json" in artifact_refs] + lead_repairs = [] + for index, decision in enumerate(receipt.get("dispositions", [])): + if decision.get("disposition") != "revise": + continue + lead_id = imported_leads[index] if index < len(imported_leads) else None + refs = [terminal_ref] + if lead_id is not None: + refs.append(artifact_refs[f"lead-attempt-{lead_id}.json"]) + lead_repairs.append({"lead_attempt_id": lead_id, + "description": decision.get("reason") or "Explicit lead revision requested.", + "evidence_refs": refs}) + if lead_repairs: + # A successful final repair is not an independently successful original. + for contribution in contributions: + if contribution["result"] == "successful": + contribution.update(result="repair", independent_success=False) + lead = receipt.get("lead", {}).get("disposition") or "not_reached" + return { + "schema_version": 1, "outcome_id": assignment["outcome_id"] if assignment is not None else f"objective-final-{run['id']}", + "kind": "final", "verdict": run["state"], "selection_mode": mode, + "observed_at": receipt.get("completed_at") or receipt.get("finished_at") or run["updated_at"], + "corrects_outcome_id": None, "summary": f"Objective terminal state: {run['state']}; lead disposition: {lead}.", + "criteria": criteria, "contributions": contributions, "lead_repairs": lead_repairs, + "evidence_refs": list(artifact_refs.values()), + } diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 22f1750..792d55b 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -892,6 +892,15 @@ def _continue_preparation( if store.remaining_wall_seconds(run_id) == 0: raise BudgetExhausted("run wall-time budget is exhausted in preflight") package, digest = self._freeze_package() + if submitted.get("trial") is not None: + from .experiment_provenance import assignment_for + from .store import git_common_dir + trial = submitted["trial"] + snapshot["experiment_spec"] = trial["experiment"] + snapshot["experiment_assignment"] = assignment_for( + trial["experiment"], trial["case_id"], trial["arm"], + project_common_dir=str(git_common_dir(Path(task["project"]["repo_path"]))), + ) version = store.complete_preparation( run_id, fencing_token, @@ -945,6 +954,7 @@ def start( idempotency_key: str, supersedes_run_id: str | None = None, *, + trial: dict[str, Any] | None = None, _internal_fake_delay: float | None = None, _internal_review_fixture: dict[str, Any] | None = None, _internal_lead_fixture: dict[str, Any] | None = None, @@ -952,6 +962,25 @@ def start( _internal_decision_fixture: dict[str, Any] | None = None, ) -> dict[str, Any]: validate_task(task, require_existing_repo=True) + if trial is not None: + from .learning import validate_experiment + from .experiment_provenance import assignment_for + from .store import git_common_dir + if not isinstance(trial, dict) or set(trial) != {"experiment", "case_id", "arm"}: + raise ContractError("trial requires an explicit experiment, case and arm") + experiment = validate_experiment(trial["experiment"]) + if experiment["schema_version"] != 2: + raise ContractError("public trials require a v2 predeclared experiment") + if (experiment["budget"]["max_cases"] > 100 + or experiment["budget"]["max_worker_invocations"] > 1000 + or experiment["budget"]["wall_seconds"] > 3600): + raise ContractError("public trial exceeds the bounded controller limits") + if ((experiment["variable"]["role"] == "reviewer" and task["workflow"] != "branch-review") + or (experiment["variable"]["role"] == "implementer" and task["workflow"] != "issue-delivery")): + raise ContractError("trial role requires a frozen review candidate or implementation baseline") + common_dir = str(git_common_dir(Path(task["project"]["repo_path"]))) + assignment_for(experiment, trial["case_id"], trial["arm"], project_common_dir=common_dir) + trial = json.loads(canonical_json({**trial, "experiment": experiment})) if _internal_fake_delay is not None and _internal_review_fixture is not None: raise ContractError("internal lifecycle fixtures are mutually exclusive") if _internal_lead_fixture is not None and _internal_fake_delay is not None: @@ -966,6 +995,8 @@ def start( "internal decision fixture requires public preflight", ) submitted = {"task": task, "supersedes_run_id": supersedes_run_id} + if trial is not None: + submitted["trial"] = trial if _internal_fake_delay is not None: submitted["_internal_fake_delay"] = _internal_fake_delay if _internal_review_fixture is not None: @@ -980,7 +1011,7 @@ def start( submitted["_internal_decision_fixture"] = _internal_decision_fixture store = self._store() try: - claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}") + claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}", objective_outcome=True) if not claim.created: return {"run_id": claim.run_id, "state": store.run(claim.run_id)["state"], "created": False} launch, error = self._continue_preparation( @@ -994,6 +1025,13 @@ def start( self._spawn_daemon(claim.run_id, version, package, digest) return {"run_id": claim.run_id, "state": "queued", "created": True} + def trial_start(self, experiment: dict[str, Any], case_id: str, arm: str, + task: dict[str, Any], idempotency_key: str, **fixtures) -> dict[str, Any]: + """Explicit one-arm controller; no automatic dispatch or promotion.""" + return self.start(task, idempotency_key, + trial={"experiment": experiment, "case_id": case_id, "arm": arm}, + **fixtures) + def _spawn_daemon(self, run_id: str, expected_version: int, package: Path, digest: str) -> int: command = [sys.executable, "-P", "-m", "devsquad.detached", "--database", str(self.database), "--artifacts", str(self.artifacts), "--run-id", run_id, "--expected-version", str(expected_version), "--package-digest", digest] # Claude's native saved-login lookup needs the login name and HOME. @@ -1233,6 +1271,7 @@ def _status_capacity(store: Store, run: dict[str, Any]) -> dict[str, Any] | None def status(self, run_id: str) -> dict[str, Any]: store = self._store() try: + store.project_final_outcome(run_id) run, attempt, handoff = store.status_snapshot(run_id) active=attempt if attempt and attempt.get("status") in {"reserved","running","cancelling","ownership_ambiguous"} else None if run["state"] == "blocked": @@ -1286,6 +1325,7 @@ def events(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any def result(self, run_id: str) -> dict[str, Any]: store = self._store() try: + store.project_final_outcome(run_id) run, artifacts = store.result_snapshot(run_id) if run["state"] not in TERMINAL_STATES: return {"run_id":run_id,"ready":False,"state":run["state"],"artifacts":[]} diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index f916328..716c5a5 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -5,6 +5,7 @@ from dataclasses import dataclass from datetime import datetime, timedelta, timezone import hashlib +from functools import wraps from importlib.resources import files import json import os @@ -17,7 +18,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 15 +SUPPORTED_SCHEMA_VERSION = 16 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -146,6 +147,15 @@ def git_common_dir(worktree: Path) -> Path: return Path(result.stdout.strip()).resolve(strict=True) +def _project_terminal(method): + @wraps(method) + def wrapped(self, run_id, *args, **kwargs): + result = method(self, run_id, *args, **kwargs) + self.project_final_outcome(run_id) + return result + return wrapped + + class Store: def __init__(self, database: Path, artifacts: Path): self.database = database @@ -238,7 +248,7 @@ def _project(self, common_dir: Path) -> str: self.connection.execute("INSERT INTO projects(id, git_common_dir, created_at) VALUES(?,?,?)", (project_id, key, _utc_now())) return project_id - def claim_start(self, worktree: Path, idempotency_key: str, submitted_request: Any, owner_id: str) -> StartClaim: + def claim_start(self, worktree: Path, idempotency_key: str, submitted_request: Any, owner_id: str, *, objective_outcome: bool = False) -> StartClaim: if not idempotency_key or not owner_id: raise ContractError("idempotency key and owner are required") encoded, digest = canonical_json(submitted_request), request_hash(submitted_request) @@ -268,6 +278,8 @@ def claim_start(self, worktree: Path, idempotency_key: str, submitted_request: A "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,1,'run.preparing','{}',?)", (run_id, now), ) + if objective_outcome: + self.connection.execute("INSERT INTO objective_outcome_jobs(run_id,requested_at) VALUES(?,?)", (run_id, now)) self.connection.execute("COMMIT") return StartClaim(run_id, project_id, digest, 1, True, 1) except Exception: @@ -540,6 +552,7 @@ def _freeze_experiment_assignment( canonical_json(snapshot), package_digest, run_version, fencing_token, recorded_at), ) + @_project_terminal def fail_preparation( self, run_id: str, @@ -606,6 +619,7 @@ def fail_preparation( self.connection.execute("ROLLBACK") raise + @_project_terminal def cancel_preparing(self, run_id: str) -> int: self.connection.execute("BEGIN IMMEDIATE") try: @@ -1591,12 +1605,47 @@ def cancel_run_decision_observations( self.connection.execute("ROLLBACK") raise + def project_final_outcome(self, run_id: str) -> dict[str, Any] | None: + """Idempotently drain a new run's durable terminal projection request.""" + from .objective_outcomes import project_outcome + job = self.connection.execute("SELECT completed_outcome_id FROM objective_outcome_jobs WHERE run_id=?", (run_id,)).fetchone() + if job is None or job["completed_outcome_id"] is not None: + return None + self.connection.execute("BEGIN") + try: + run = self.run(run_id) + if run["state"] not in TERMINAL_STATES: + self.connection.execute("COMMIT") + return None + refs, documents = {}, {} + for artifact in self.artifacts_for_run(run_id): + content = Path(artifact["path"]).read_bytes() + if hashlib.sha256(content).hexdigest() != artifact["sha256"]: + raise ConflictError("objective projection artifact integrity is invalid") + refs[artifact["name"]] = f"artifact:{artifact['id']}:{artifact['sha256']}" + if artifact["name"] in {"receipt.json", "result-receipt.json"}: + documents[artifact["name"]] = json.loads(content) + receipt = documents.get("receipt.json", documents.get("result-receipt.json")) + if not isinstance(receipt, dict): + raise ConflictError("objective projection requires a durable terminal receipt") + row = self.connection.execute("SELECT assignment_json FROM experiment_assignments WHERE run_id=?", (run_id,)).fetchone() + assignment = json.loads(row["assignment_json"]) if row else None + outcome = project_outcome(run, receipt, self.attempts_for_run(run_id), refs, assignment) + self.connection.execute("COMMIT") + except Exception: + self.connection.execute("ROLLBACK") + raise + result = self.record_outcome(run_id, outcome, _objective_projection=True) + self.connection.execute("UPDATE objective_outcome_jobs SET completed_outcome_id=? WHERE run_id=? AND completed_outcome_id IS NULL", (outcome["outcome_id"], run_id)) + return result + def record_outcome( self, run_id: str, outcome: dict[str, Any], *, now: datetime | None = None, + _objective_projection: bool = False, ) -> dict[str, Any]: """Append one replay-safe final outcome or late correction.""" from .learning import validate_outcome @@ -1646,6 +1695,8 @@ def record_outcome( and normalized["selection_mode"] != expected_selection_mode): raise ConflictError("outcome selection mode does not match frozen routing") if normalized["kind"] == "final": + if not _objective_projection and self.connection.execute("SELECT 1 FROM objective_outcome_jobs WHERE run_id=?", (run_id,)).fetchone(): + raise ConflictError("public final outcomes are objective projections; use an explicit late correction") if normalized["verdict"] != run["state"]: raise ConflictError("final outcome verdict does not match run state") else: @@ -1723,6 +1774,9 @@ def record_outcome( raise def outcomes_for_run(self, run_id: str) -> list[dict[str, Any]]: + # Repair only the requested run. Corrupt evidence in another pending + # projection must not prevent opening the ledger or observing good runs. + self.project_final_outcome(run_id) if not self.connection.execute( "SELECT 1 FROM runs WHERE id=?", (run_id,), ).fetchone(): @@ -1744,7 +1798,7 @@ def outcomes_for_run(self, run_id: str) -> list[dict[str, Any]]: def learning_report( self, project: Path, *, now: datetime | None = None, ) -> dict[str, Any]: - """Build a read-only project comparison with explicit missingness.""" + """Repair pending public projections and compare with explicit missingness.""" from .learning import build_comparison_report common_dir = git_common_dir(project) @@ -1766,6 +1820,8 @@ def learning_report( (project_id,), ) ] + for terminal_run in terminal_runs: + self.project_final_outcome(terminal_run["run_id"]) outcome_records = [ {"run_id": row["run_id"], "outcome": json.loads(row["payload_json"])} for row in self.connection.execute( @@ -1818,6 +1874,10 @@ def evaluate_learning_experiment( spec_sha256 = hashlib.sha256(spec_json.encode()).hexdigest() current = _authoritative_now(now) common_dir = git_common_dir(project_path) + for assigned in self.connection.execute( + "SELECT run_id FROM experiment_assignments WHERE experiment_id=?", + (spec["experiment_id"],)).fetchall(): + self.project_final_outcome(assigned["run_id"]) self.connection.execute("BEGIN IMMEDIATE") try: existing = saved_evaluation(self.connection, spec["experiment_id"]) @@ -2885,6 +2945,28 @@ def worker_invocations(self, run_id: str) -> int: (run_id,), ).fetchone()[0] + def _enforce_experiment_budget(self, run_id: str) -> None: + """Shared experiment fence; caller holds the reservation write lock.""" + row = self.connection.execute( + "SELECT s.spec_json,s.recorded_at FROM experiment_assignments a " + "JOIN experiment_specs s ON s.experiment_id=a.experiment_id WHERE a.run_id=?", + (run_id,), + ).fetchone() + if row is None: + return + spec = json.loads(row["spec_json"]) + # Count every durable reservation, including failed/fallback/revision + # slots and concurrent not-yet-launched work. Never trust caller counters. + consumed = self.connection.execute( + "SELECT COUNT(*) FROM attempts t JOIN experiment_assignments a ON a.run_id=t.run_id " + "WHERE a.experiment_id=?", (spec["experiment_id"],), + ).fetchone()[0] + if consumed >= spec["budget"]["max_worker_invocations"]: + raise BudgetExhausted("experiment worker reservation budget is exhausted") + elapsed = (_authoritative_now() - datetime.fromisoformat(row["recorded_at"])).total_seconds() + if elapsed >= spec["budget"]["wall_seconds"]: + raise BudgetExhausted("experiment wall-time budget is exhausted") + def reserve_attempt( self, run_id: str, @@ -2926,6 +3008,7 @@ def reserve_attempt( package_digest=package_digest, ) self._enforce_attempt_budget(run_id, run) + self._enforce_experiment_budget(run_id) if account_pool_id is not None: if not isinstance(account_pool_id, str) or not account_pool_id: raise ContractError("attempt account pool is invalid") @@ -3077,6 +3160,7 @@ def request_cancel(self, run_id: str) -> tuple[int, dict[str, Any] | None]: self.connection.execute("ROLLBACK") raise + @_project_terminal def finish_attempt(self, run_id: str, attempt_token: str, terminal_state: str, payload: Any) -> int: if terminal_state not in TERMINAL_STATES: raise ContractError("invalid terminal state") @@ -3268,6 +3352,7 @@ def _record_prepared_output( raise ConflictError("durable output conflicts with its prior import") return version + @_project_terminal def commit_durable_import(self, run_id: str, attempt_token: str, artifacts: list[dict[str, Any]], metadata: Any, terminal_state: str, payload: Any) -> str: """Atomically import one durable receipt, or observe its prior import. @@ -3808,6 +3893,7 @@ def block_recovery(self, run_id: str, attempt_token: str, reason: str, *, releas self.connection.execute("ROLLBACK") raise + @_project_terminal def recover_unstarted_attempt( self, run_id: str, attempt_token: str, reason: str, ) -> tuple[int, str]: @@ -3942,6 +4028,7 @@ def request_recovery_cancel(self, run_id: str, attempt_token: str) -> int: self.connection.execute("ROLLBACK") raise + @_project_terminal def finish_recovery_cancel( self, run_id: str, attempt_token: str, reason: str, ) -> int: @@ -4897,6 +4984,7 @@ def requeue_delivery_revision( self.connection.execute("ROLLBACK") raise + @_project_terminal def complete_handoff_terminal( self, run_id: str, @@ -4978,6 +5066,7 @@ def complete_handoff_terminal( self.connection.execute("ROLLBACK") raise + @_project_terminal def cancel_host_wait( self, run_id: str, @@ -5051,6 +5140,7 @@ def cancel_host_wait( self.connection.execute("ROLLBACK") raise + @_project_terminal def fail_queued_budget( self, run_id: str, @@ -5107,6 +5197,7 @@ def fail_queued_budget( self.connection.execute("ROLLBACK") raise + @_project_terminal def cancel_queued(self, run_id: str) -> int: self.connection.execute("BEGIN IMMEDIATE") try: @@ -5129,6 +5220,7 @@ def cancel_queued(self, run_id: str) -> int: except Exception: self.connection.execute("ROLLBACK"); raise + @_project_terminal def cancel_launching(self, run_id: str) -> int: self.connection.execute("BEGIN IMMEDIATE") try: diff --git a/test/core/experiment_runtime_fixture.py b/test/core/experiment_runtime_fixture.py index 21b0054..2ef8489 100644 --- a/test/core/experiment_runtime_fixture.py +++ b/test/core/experiment_runtime_fixture.py @@ -1,6 +1,6 @@ -"""Real offline experiment runs with a test-only preflight assignment seam. +"""Real offline experiment runs through the explicit public trial controller. -This is not a public experiment controller or native-provider quality proof. +This is not native-provider quality proof. Profiles and reviews are explicit fixtures, but preparation, worker processes, checks, host decisions and outcomes use the actual Service/Store workflow. """ @@ -9,7 +9,6 @@ from contextlib import ExitStack import copy -from datetime import datetime, timezone from pathlib import Path import subprocess import sys @@ -17,10 +16,9 @@ from unittest.mock import patch from test_experiment_provenance import digest, execution_digest -from test_learning import experimental_final from test_lifecycle import profile, review_task, routing_policy -from devsquad.experiment_provenance import assignment_for, paired_input_identity, selected_execution_fingerprint +from devsquad.experiment_provenance import paired_input_identity, selected_execution_fingerprint from devsquad.codex_review_worker import freeze_codex_reviewer from devsquad.router import load_routing from devsquad.service import Service @@ -202,18 +200,7 @@ def wait(self, run_id, *, candidate_ready=True): raise AssertionError(f"saved-run fixture did not reach a gate: {self.service.status(run_id)}") def run_arm(self, case_id, arm, *, no_attempt=False): - original = self.service._resolve_snapshot - - def predeclared_assignment(*args, **kwargs): - snapshot = original(*args, **kwargs) - snapshot["experiment_spec"] = copy.deepcopy(self.spec) - snapshot["experiment_assignment"] = assignment_for( - self.spec, case_id, arm, project_common_dir=self.common, - ) - return snapshot - with ExitStack() as stack: - stack.enter_context(patch.object(self.service, "_resolve_snapshot", side_effect=predeclared_assignment)) if no_attempt: stack.enter_context(patch.object(self.service, "_spawn_daemon", return_value=0)) fixture_args = {} @@ -225,8 +212,8 @@ def predeclared_assignment(*args, **kwargs): } if not self.native_review: fixture_args["_internal_review_fixture"] = {"verdict": "clean", "summary": "Fixture review of the frozen candidate.", "findings": []} - started = self.service.start( - self.task(case_id, arm), self.outcome_id(case_id, arm), + started = self.service.trial_start( + self.spec, case_id, arm, self.task(case_id, arm), self.outcome_id(case_id, arm), **fixture_args, ) run_id = started["run_id"] @@ -258,10 +245,6 @@ def predeclared_assignment(*args, **kwargs): verdict = "succeeded" if (arm == "candidate") == self.candidate_succeeds else "failed" if completed["state"] != verdict: raise AssertionError(f"public fixture disposition failed: {completed}") - outcome = experimental_final(self.outcome_id(case_id, arm), verdict) - outcome["observed_at"] = datetime.now(timezone.utc).isoformat() - outcome["evidence_refs"] = ["receipt.json", "result-receipt.json"] - self.service.outcome_add(run_id, outcome) return run_id def run_all(self, *, skip=None, no_attempt=None): diff --git a/test/core/test_cli.py b/test/core/test_cli.py index 48b79c8..aeeea52 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -74,6 +74,24 @@ def test_start_returns_exact_envelope_and_forwards_inputs(self): service.start.assert_called_once_with({"schema_version": 1}, "key-1", "old-run") service.status.assert_not_called() + def test_trial_dispatch_is_explicit_and_preserves_the_predeclaration(self): + experiment = {"schema_version": 2, "experiment_id": "explicit-fixture"} + path = self.root / "trial-experiment.json" + path.write_text(json.dumps(experiment)) + service = mock.Mock() + response = {"run_id": "trial-run", "state": "queued", "created": True} + service.trial_start.return_value = response + code, payload, stderr = self.invoke([ + "trial", "--experiment", str(path), "--case", "held-out-case", "--arm", "candidate", + "--task-file", str(self.task_file), "--idempotency-key", "explicit-trial-key", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.trial_start.assert_called_once_with(experiment, "held-out-case", "candidate", {"schema_version": 1}, "explicit-trial-key") + service.start.assert_not_called() + service.status.assert_not_called() + def test_review_dry_run_prepares_a_managed_task_without_starting(self): identity = { "harness": "codex", "harness_version": "codex fixture", @@ -720,7 +738,7 @@ def test_installed_wheel_contains_and_applies_current_migrations(self): from pathlib import Path import sqlite3 import sys -from devsquad.store import Store +from devsquad.store import Store, SUPPORTED_SCHEMA_VERSION root = Path(sys.argv[1]) root.mkdir(parents=True) @@ -734,8 +752,9 @@ def test_installed_wheel_contains_and_applies_current_migrations(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 15 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == SUPPORTED_SCHEMA_VERSION assert migrations.joinpath("015_experiment_evaluation_revisions.sql").is_file() + assert migrations.joinpath("016_objective_outcome_jobs.sql").is_file() attempt_columns = {row[1] for row in store.connection.execute("PRAGMA table_info(attempts)")} assert {"role", "account_pool_id", "profile_id", "profile_index"} <= attempt_columns columns = {row[1] for row in store.connection.execute("PRAGMA table_info(runs)")} diff --git a/test/core/test_experiment_eligibility.py b/test/core/test_experiment_eligibility.py index b3e2ea5..ecf40fa 100644 --- a/test/core/test_experiment_eligibility.py +++ b/test/core/test_experiment_eligibility.py @@ -20,7 +20,7 @@ from devsquad.contracts import ContractError from devsquad.learning import evaluate_experiment from devsquad.service import Service -from devsquad.store import ConflictError, Store, canonical_json +from devsquad.store import ConflictError, Store, canonical_json, SUPPORTED_SCHEMA_VERSION class ExperimentEligibilityTest(unittest.TestCase): @@ -377,7 +377,7 @@ def test_upgraded_reused_outcome_history_remains_readable_but_cannot_authorize_n store = Store(database, runtime / "artifacts") self.addCleanup(store.close) original = dict(store.connection.execute("SELECT * FROM experiments").fetchone()) - self.assertEqual(store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) + self.assertEqual(store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) report = service.learning_report(str(repo)) self.assertEqual(report["sample_size"], 0) proposal = service.learning_propose(str(repo))["proposal"] diff --git a/test/core/test_install_core.py b/test/core/test_install_core.py index fdb657f..65766a4 100644 --- a/test/core/test_install_core.py +++ b/test/core/test_install_core.py @@ -16,6 +16,7 @@ from unittest.mock import patch from devsquad_test_fixtures import branch_review_routing_documents +from devsquad.store import SUPPORTED_SCHEMA_VERSION ROOT = Path(__file__).resolve().parents[2] @@ -519,7 +520,7 @@ def test_schema_13_update_defers_for_active_and_recoverable_old_runs(self): self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 13) self.assertTrue(self.cli_json(launcher, "result", queued["run_id"], "--runtime-dir", str(runtime))["data"]["ready"]) with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: - self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) + self.assertEqual(connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) finally: # Use the old package explicitly even if a later assertion fails. cleanup_script = "from pathlib import Path; import sys; from devsquad.service import Service; s=Service(Path(sys.argv[1])); [s.cancel(r) for r in sys.argv[2:]]" @@ -547,7 +548,7 @@ def test_failed_selector_activation_never_advances_the_old_ledger(self): temporary.symlink_to("new-release") with patch("devsquad.release_activation.os.replace", side_effect=OSError("injected activation failure")): with self.assertRaisesRegex(OSError, "injected activation failure"): - activate_release(temporary, selector, runtime, supported_schema_version=15) + activate_release(temporary, selector, runtime, supported_schema_version=SUPPORTED_SCHEMA_VERSION) self.assertEqual(os.readlink(selector), "old-release") with closing(sqlite3.connect(database, timeout=1)) as connection: connection.execute("BEGIN EXCLUSIVE") @@ -592,7 +593,7 @@ def test_preopened_schema13_store_cannot_admit_work_after_new_schema_commit(self self.install() store = Store(runtime / "state.sqlite3", runtime / "artifacts") try: - self.assertEqual(store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) + self.assertEqual(store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) finally: store.close() output, stderr = process.communicate("admit\n", timeout=8) diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index b743c82..8ecc229 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -174,7 +174,7 @@ def seed_experiment( candidate_succeeds=True, ): # Positive authority comes from real workers and public completion, - # never SQL-terminalized empty runs. The assignment seam remains R5. + # never SQL-terminalized empty runs or manually imported final outcomes. from experiment_runtime_fixture import ExperimentRuntimeFixture fixture = ExperimentRuntimeFixture( diff --git a/test/core/test_objective_outcomes.py b/test/core/test_objective_outcomes.py new file mode 100644 index 0000000..799d131 --- /dev/null +++ b/test/core/test_objective_outcomes.py @@ -0,0 +1,143 @@ +"""Public terminal-origin regressions: no manually imported final outcomes.""" +import copy +from datetime import datetime, timezone +from pathlib import Path +import sys +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path[:0] = [str(ROOT / "plugin/core/src"), str(ROOT / "test/core")] + +from devsquad.store import ConflictError, Store +import test_review_runtime as review_fixtures + + +class ObjectiveOutcomeTest(unittest.TestCase): + def setUp(self): + self.fixture = review_fixtures.DurableBranchReviewTest() + self.fixture.setUp() + self.addCleanup(self.fixture.doCleanups) + self.service = self.fixture.service + self.fixture.task["checks"][0]["argv"] = [sys.executable, "-c", "print('verified')"] + + def outcome(self, run_id): + store = self.service._store() + try: + rows = store.outcomes_for_run(run_id) + self.assertEqual(len(rows), 1, "public terminal run must project exactly one final outcome") + return rows[0]["outcome"] + finally: + store.close() + + def accept(self, run_id, waiting): + claimed = self.service.handoff_claim(run_id, waiting["version"], "objective-fixture-host") + decision = self.fixture.decision(claimed["handoff"]["packet"], "objective-accept", "accept", "Verified objective fixture evidence.") + return self.service.handoff_complete(run_id, claimed["claim"], decision) + + def test_preparation_failure_projects_missing_evidence_truthfully(self): + task = copy.deepcopy(self.fixture.task) + task["project"]["target_ref"] = "nonexistent-objective-target" + started = self.service.start(task, "objective-preparation-failure") + self.assertEqual(started["state"], "failed") + outcome = self.outcome(started["run_id"]) + self.assertEqual(outcome["verdict"], "failed") + self.assertEqual(outcome["contributions"], []) + self.assertTrue(all(c["status"] == "unknown" and not c["evidence_refs"] for c in outcome["criteria"])) + + def test_prelaunch_cancel_has_no_completed_exposure(self): + with mock.patch.object(self.service, "_spawn_daemon", return_value=0): + started = self.service.start(self.fixture.task, "objective-prelaunch-cancel", _internal_review_fixture=self.fixture.fixture) + self.service.cancel(started["run_id"]) + outcome = self.outcome(started["run_id"]) + self.assertEqual(outcome["verdict"], "cancelled") + self.assertEqual(outcome["contributions"], []) + self.assertEqual(self.outcome(started["run_id"]), outcome) + + def test_worker_failure_is_not_independent_success(self): + self.fixture.task["checks"][0]["cwd"] = "missing-check-directory" + started = self.service.start(self.fixture.task, "objective-worker-failure", _internal_review_fixture=self.fixture.fixture) + self.fixture.wait_state(started["run_id"], {"failed"}) + outcome = self.outcome(started["run_id"]) + self.assertEqual(outcome["verdict"], "failed") + self.assertTrue(outcome["contributions"]) + self.assertTrue(all(c["result"] == "failed" and not c["independent_success"] for c in outcome["contributions"])) + + def test_host_completion_and_late_correction_are_append_only(self): + run_id, waiting = self.fixture.start_waiting("objective-host") + self.assertEqual(self.accept(run_id, waiting)["state"], "succeeded") + final = self.outcome(run_id) + self.assertEqual(final["verdict"], "succeeded") + self.assertTrue(final["contributions"]) + correction = {**final, "outcome_id": "objective-late-correction", "kind": "late_correction", "verdict": "escaped_defect", + "corrects_outcome_id": final["outcome_id"], "observed_at": datetime.now(timezone.utc).isoformat(), "summary": "Explicit later escaped-defect evidence."} + self.service.outcome_add(run_id, correction) + self.service.result(run_id) + store = self.service._store() + try: + rows = store.outcomes_for_run(run_id) + self.assertEqual(len(rows), 2) + self.assertEqual(rows[0]["outcome"], final) + finally: + store.close() + + def test_headless_completion_projects_without_manual_import(self): + self.fixture.configure_fixture_headless() + started = self.service.start(self.fixture.task, "objective-headless", _internal_review_fixture=self.fixture.fixture, + _internal_lead_fixture={"disposition": "accept", "reason": "Verified fixture evidence."}) + completed = self.fixture.wait_state(started["run_id"], {"succeeded", "failed"}) + self.assertEqual(completed["state"], "succeeded") + outcome = self.outcome(started["run_id"]) + self.assertEqual({c["role"] for c in outcome["contributions"]}, {"reviewer", "lead"}) + + def test_projection_crash_replays_one_outcome_after_terminal_commit(self): + run_id, waiting = self.fixture.start_waiting("objective-crash") + with mock.patch.object(Store, "project_final_outcome", create=True, side_effect=RuntimeError("projection crash")): + with self.assertRaisesRegex(RuntimeError, "projection crash"): + self.accept(run_id, waiting) + self.assertEqual(self.service.status(run_id)["state"], "succeeded") + final = self.outcome(run_id) + self.service.result(run_id) + self.assertEqual(self.outcome(run_id), final) + + def test_repaired_failed_attempt_never_gets_independent_credit(self): + self.fixture.configure_reviewer_fallback() + run_id, waiting = self.fixture.start_waiting("objective-repaired") + self.accept(run_id, waiting) + contributions = self.outcome(run_id)["contributions"] + self.assertEqual([c["result"] for c in contributions], ["failed", "repair"]) + self.assertTrue(all(not c["independent_success"] for c in contributions)) + + def test_report_repairs_a_projection_crash_without_manual_outcome_import(self): + run_id, waiting = self.fixture.start_waiting("objective-report-crash") + with mock.patch.object(Store, "project_final_outcome", side_effect=RuntimeError("projection crash")): + with self.assertRaisesRegex(RuntimeError, "projection crash"): + self.accept(run_id, waiting) + report = self.service.learning_report(self.fixture.repo) + self.assertEqual(report["sample_size"], 1) + self.assertEqual(report["missingness"]["terminal_runs_without_final_outcome"], 0) + final = self.outcome(run_id) + with self.assertRaisesRegex(ConflictError, "objective projections"): + self.service.outcome_add(run_id, {**final, "outcome_id": "manual-replacement-final"}) + + def test_corrupt_pending_projection_does_not_block_other_runs(self): + run_id, waiting = self.fixture.start_waiting("objective-corrupt-pending") + with mock.patch.object(Store, "project_final_outcome", side_effect=RuntimeError("projection crash")): + with self.assertRaises(RuntimeError): + self.accept(run_id, waiting) + store = self.service._store() + try: + artifact = next(a for a in store.artifacts_for_run(run_id) if a["name"] == "receipt.json") + Path(artifact["path"]).write_bytes(b"corrupt fixture receipt") + finally: + store.close() + with self.assertRaisesRegex(ConflictError, "integrity"): + self.service.status(run_id) + good_id, good_waiting = self.fixture.start_waiting("objective-unrelated-good") + self.assertEqual(self.accept(good_id, good_waiting)["state"], "succeeded") + self.assertEqual(self.service.status(good_id)["state"], "succeeded") + self.assertEqual(self.outcome(good_id)["verdict"], "succeeded") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_public_trials.py b/test/core/test_public_trials.py new file mode 100644 index 0000000..ef4c4bd --- /dev/null +++ b/test/core/test_public_trials.py @@ -0,0 +1,184 @@ +"""Public opt-in trials: declaration, shared budgets and lifecycle chain.""" +import copy +import json +from pathlib import Path +import tempfile +import threading +import time +import unittest +from unittest.mock import patch + +from experiment_runtime_fixture import ExperimentRuntimeFixture +import test_lifecycle as lifecycle_fixtures +from devsquad.contracts import ContractError + + +class PublicTrialTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="devsquad-public-trial-") + self.addCleanup(self.temporary.cleanup) + self.fixture = ExperimentRuntimeFixture(Path(self.temporary.name)) + self.addCleanup(self.fixture.close) + + def start(self, case="eval-1", arm="candidate", *, fixture=None): + fixture = fixture or self.fixture + started = fixture.service.trial_start( + fixture.spec, case, arm, fixture.task(case, arm), f"public-trial-{case}-{arm}", + _internal_review_fixture={"verdict": "clean", "summary": "Explicit offline trial review.", "findings": []}, + ) + fixture.runs[(case, arm)] = started["run_id"] + return started + + def test_opt_in_declaration_is_immutable_and_exact_replay_does_not_launch_again(self): + with patch.object(self.fixture.service, "_spawn_daemon", return_value=0) as spawn: + first = self.start() + replay = self.start() + self.assertEqual(replay["run_id"], first["run_id"]) + self.assertFalse(replay["created"]) + spawn.assert_called_once() + original = copy.deepcopy(self.fixture.spec) + self.fixture.spec["hypothesis"] = "Changed after one arm was frozen." + changed = self.start(arm="control") + self.assertEqual(changed["state"], "failed") + store = self.fixture.store() + try: + self.assertEqual(json.loads(store.connection.execute("SELECT spec_json FROM experiment_specs").fetchone()[0]), original) + self.assertEqual(store.attempts_for_run(changed["run_id"]), []) + self.assertEqual(len(store.outcomes_for_run(changed["run_id"])), 1) + finally: + store.close() + + def test_shared_experiment_budget_is_not_replenished_for_another_arm(self): + self.fixture.spec["budget"]["max_worker_invocations"] = 1 + self.fixture.run_arm("eval-1", "control") + second = self.start() + self.assertEqual(self.fixture.wait(second["run_id"])["state"], "failed") + store = self.fixture.store() + try: + self.assertEqual(store.attempts_for_run(second["run_id"]), []) + self.assertEqual(store.outcomes_for_run(second["run_id"])[0]["outcome"]["contributions"], []) + self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM attempts").fetchone()[0], 1) + finally: + store.close() + evaluation = self.fixture.service.policy_evaluate(self.fixture.spec)["evaluation"] + self.assertEqual(evaluation["metrics"]["evaluation"]["available_pairs"], 0) + + def test_concurrent_arms_cannot_overbook_the_same_trial_budget(self): + self.fixture.spec["budget"]["max_worker_invocations"] = 1 + with patch.object(self.fixture.service, "_spawn_daemon", return_value=0): + runs = [self.start(arm=arm)["run_id"] for arm in ("control", "candidate")] + barrier, errors = threading.Barrier(2), [] + def resume(run_id): + try: + barrier.wait(timeout=5) + self.fixture.service.resume(run_id) + except Exception as exc: + errors.append(exc) + workers = [threading.Thread(target=resume, args=(run_id,)) for run_id in runs] + for worker in workers: + worker.start() + for worker in workers: + worker.join(timeout=10) + self.assertFalse(worker.is_alive()) + self.assertEqual(errors, []) + states = [self.fixture.wait(run_id)["state"] for run_id in runs] + self.assertCountEqual(states, ["awaiting_host", "failed"]) + store = self.fixture.store() + try: + self.assertEqual(store.connection.execute("SELECT COUNT(*) FROM attempts").fetchone()[0], 1) + finally: + store.close() + + def test_fallback_cannot_spend_a_second_slot_after_the_trial_budget(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "fallback", with_fallback=True) + self.addCleanup(fixture.close) + fixture.spec["budget"]["max_worker_invocations"] = 1 + started = self.start(fixture=fixture) + self.assertEqual(fixture.wait(started["run_id"])["state"], "failed") + store = fixture.store() + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual(len(attempts), 1) + self.assertEqual(attempts[0]["profile_index"], 0) + final = store.outcomes_for_run(started["run_id"])[0]["outcome"] + self.assertEqual([c["result"] for c in final["contributions"]], ["failed"]) + finally: + store.close() + + def test_experiment_deadline_blocks_later_launch_without_invented_exposure(self): + self.fixture.spec["budget"]["wall_seconds"] = 1 + with patch.object(self.fixture.service, "_spawn_daemon", return_value=0): + started = self.start() + time.sleep(1.05) + self.fixture.service.resume(started["run_id"]) + self.assertEqual(self.fixture.wait(started["run_id"])["state"], "failed") + store = self.fixture.store() + try: + self.assertEqual(store.attempts_for_run(started["run_id"]), []) + self.assertEqual(store.outcomes_for_run(started["run_id"])[0]["outcome"]["contributions"], []) + finally: + store.close() + + def test_public_controller_rejects_delivery_reviewer_and_unbounded_requests(self): + task = self.fixture.task("eval-1", "control") + task.update(workflow="issue-delivery") + task["scope"]["write_paths"] = ["README"] + with self.assertRaisesRegex(ContractError, "frozen review"): + self.fixture.service.trial_start(self.fixture.spec, "eval-1", "control", task, "invalid-delivery-reviewer") + experiment = copy.deepcopy(self.fixture.spec) + experiment["budget"]["wall_seconds"] = 3601 + with self.assertRaisesRegex(ContractError, "bounded controller"): + self.fixture.service.trial_start(experiment, "eval-1", "control", self.fixture.task("eval-1", "control"), "unbounded-trial") + + def test_public_outcomes_evaluate_qualify_promote_new_run_and_roll_back(self): + service = self.fixture.service + service.profile_binding_bootstrap({"template": lifecycle_fixtures.lifecycle_template(update_mode="reviewed"), + "profile": self.fixture.profiles["control"], "version": 7}) + self.fixture.run_all() + self.assertEqual(service.learning_report(self.fixture.repo)["sample_size"], 4) + evaluation = service.policy_evaluate(self.fixture.spec) + self.assertTrue(evaluation["eligibility"]["eligible"]) + helper = lifecycle_fixtures.ProfileLifecycleTest() + helper.candidate = self.fixture.profiles["candidate"] + qualification = helper.qualification(evaluation) + self.assertEqual(service.profile_qualification_add(qualification)["gate_failures"], []) + promoted = service.profile_binding_change(helper.promotion("public-chain-promote")) + self.assertEqual(promoted["receipt"]["to"]["binding_version"], 8) + # Normal automatic routing of a NEW run observes the promoted alias; + # completed experimental runs retain their exact prelaunch bindings. + task = self.fixture.task("hold-1", "candidate") + task["routing"].pop("overrides") + with patch.object(service, "_spawn_daemon", return_value=0): + automatic = service.start(task, "public-promoted-new-run", + _internal_review_fixture={"verdict": "clean", "summary": "New-run binding verification.", "findings": []}) + try: + store = self.fixture.store() + try: + snapshot = json.loads(store.run(automatic["run_id"])["mutable_snapshot"]) + self.assertEqual(snapshot["routing"]["roles"]["reviewer"]["selected"]["profile_id"], "profile-b") + self.assertEqual(snapshot["routing"]["roles"]["reviewer"]["selected"]["binding"]["version"], 8) + old = json.loads(store.run(self.fixture.runs[("eval-1", "control")])["mutable_snapshot"]) + self.assertEqual(old["routing"]["roles"]["reviewer"]["selected"]["profile_id"], "profile-a") + finally: + store.close() + finally: + service.cancel(automatic["run_id"]) + regression = ExperimentRuntimeFixture(self.fixture.root / "regression", service=service, repo=self.fixture.repo, + experiment_id="public-post-promotion-regression", candidate_succeeds=False) + self.addCleanup(regression.close) + regression.run_all() + evaluated = service.policy_evaluate(regression.spec) + self.assertEqual(evaluated["evaluation"]["verdict"], "no_change") + rollback = {"schema_version": 1, "decision_id": "public-chain-rollback", "action": "rollback", "alias": "review.deep", + "expected_binding_version": 8, "qualification_id": None, + "rollback_target": {"profile_id": "profile-a", "binding_version": 7}, + "experiment_id": regression.spec["experiment_id"], "evaluation_sha256": evaluated["evaluation_sha256"], + "actor": "human", "reason": "Predeclared evaluation and held-out regression favor the prior incumbent.", + "evidence_refs": ["evaluation.json"]} + reverted = service.profile_binding_change(rollback) + self.assertEqual(reverted["receipt"]["to"]["binding_version"], 9) + self.assertEqual(service.profile_binding_status("review.deep")["binding"]["profile_id"], "profile-a") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_service.py b/test/core/test_service.py index c3ebb0c..8a9420f 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -402,7 +402,7 @@ def test_capacity_observation_drives_preflight_and_status_evidence(self): self.assertEqual(current["status"], "available") self.assertEqual(current["windows"][0]["window_id"], "short") - def test_outcome_add_and_project_report_share_saved_ledger(self): + def test_objective_outcome_and_project_report_share_saved_ledger(self): started = self.service.start( self.task, "learning-report", _internal_fake_delay=.01, ) @@ -410,12 +410,12 @@ def test_outcome_add_and_project_report_share_saved_ledger(self): now = datetime.now(timezone.utc) recorded = self.service.outcome_add(started["run_id"], { "schema_version": 1, - "outcome_id": "service-final-outcome", - "kind": "final", - "verdict": "succeeded", + "outcome_id": "service-explicit-correction", + "kind": "late_correction", + "verdict": "corrected", "selection_mode": "automatic", "observed_at": now.isoformat(), - "corrects_outcome_id": None, + "corrects_outcome_id": f"objective-final-{started['run_id']}", "summary": "The saved fixture run completed successfully.", "criteria": [], "contributions": [], @@ -427,7 +427,7 @@ def test_outcome_add_and_project_report_share_saved_ledger(self): self.assertEqual(report["sample_size"], 1) self.assertEqual(report["terminal_run_count"], 1) self.assertEqual(report["final_successes"], 1) - self.assertEqual(report["missingness"]["finals_without_contributions"], 1) + self.assertEqual(report["missingness"]["finals_without_contributions"], 0) proposed = self.service.learning_propose(self.repo) self.assertEqual(proposed["proposal"]["verdict"], "no_change") self.assertFalse(proposed["proposal"]["active_policy_changed"]) diff --git a/test/core/test_store.py b/test/core/test_store.py index d89e5c2..03ebb73 100644 --- a/test/core/test_store.py +++ b/test/core/test_store.py @@ -13,7 +13,7 @@ sys.path.insert(0, str(ROOT / "plugin/core/src")) from devsquad.contracts import BudgetExhausted, ContractError -from devsquad.store import ConflictError, SchemaVersionError, Store, git_common_dir +from devsquad.store import ConflictError, SchemaVersionError, Store, git_common_dir, SUPPORTED_SCHEMA_VERSION class StoreTest(unittest.TestCase): @@ -482,8 +482,8 @@ def test_wall_budget_counts_preflight_and_prior_attempts_cumulatively(self): ) def test_migration_records_version_and_refuses_newer_database(self): - self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) - self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(16,'future')") + self.assertEqual(self.store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) + self.store.connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,'future')", (SUPPORTED_SCHEMA_VERSION + 1,)) self.store.close() with self.assertRaises(SchemaVersionError): Store(self.database, self.artifacts) @@ -498,7 +498,7 @@ def test_version_one_fixture_migrates_to_current(self): connection.commit(); connection.close() upgraded = Store(old_db, self.root / "old-artifacts") self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 15) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) self.assertTrue(upgraded.connection.execute("SELECT 1 FROM sqlite_master WHERE name='attempts'").fetchone()) attempt_columns = { row[1] for row in upgraded.connection.execute("PRAGMA table_info(attempts)") @@ -515,7 +515,7 @@ def test_version_three_fixture_adds_run_snapshot_columns(self): connection.execute("INSERT INTO schema_migrations(version,applied_at) VALUES(?,?)",(version,"fixture")) connection.commit(); connection.close() upgraded=Store(old_db,self.root/"v3-artifacts"); self.addCleanup(upgraded.close) - self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0],15) + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) columns={row[1] for row in upgraded.connection.execute("PRAGMA table_info(runs)")} self.assertTrue({"package_path","package_digest","supersedes_run_id"} <= columns) From 14b3a3910972484074ed52d99f70123944e56fd0 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 14:10:37 -0700 Subject: [PATCH 175/197] WIP checkpoint: R5 schema-16 regression expectations; preserve failed affected gate (2026-10-02 14:10) --- docs/plans/engineering-team/RESUME.md | 8 +++++--- .../R5-public-integration-partial-2026-10-02.json | 5 +++-- test/core/test_capacity.py | 2 +- test/core/test_decision_store.py | 2 +- test/core/test_experiment_assignment_store.py | 2 +- test/core/test_handoff_store.py | 2 +- test/core/test_learning.py | 2 +- test/core/test_lifecycle.py | 2 +- 8 files changed, 14 insertions(+), 11 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 667a73f..85e2b5b 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -50,9 +50,11 @@ in 61.236s (before the added deadline test/lead-repair attribution refinement). Public concurrent-budget and promotion → new-run binding → held-out regression → rollback tests pass. See `evidence/R5-public-integration-partial-2026-10-02.json`. -Next: finish the frozen learning/lifecycle affected gate currently running -(exec session 99417; do not start a second copy or edit source/tests during -it). Then full core gate with explicit offline Python 3.12 build interpreter, +The frozen learning/lifecycle affected gate finished: **71 tests/233.881s**, +two stale schema-15 assertion failures, no behavioral errors. Those expectations +now match schema 16; keep the failed gate in partial evidence. No test process +remains active at this checkpoint. Next: recheck the two assertions, then the +full core gate with explicit offline Python 3.12 build interpreter, independent exact-patch review, schema-16 safe installer/installed transport recheck and R5 closure. Current production installation/ledger is still the accepted R4 **schema 15**; never open it with source Service while schema-16 diff --git a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json index 58ccbcf..0a70205 100644 --- a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json @@ -11,7 +11,7 @@ {"requirement": "public predeclared assignments and automatic finals, not preparation monkeypatch or manual import", "proof": "ExperimentRuntimeFixture now calls Service.trial_start; public reviewer and implementer pair tests", "result": "focused_pass"}, {"requirement": "one shared experiment reservation cap across concurrent arms and fallback", "proof": "test_public_trials actual concurrent public resumes and fallback failure", "result": "focused_pass"}, {"requirement": "evaluate, qualify, reviewed promotion for a new run, held-out regression and rollback", "proof": "test_public_trials complete Service API chain with actual offline worker processes", "result": "focused_pass"}, - {"requirement": "declaration-time experiment wall deadline and precise headless lead-repair attribution", "proof": "added regression/refinement after initial 88-test gate", "result": "affected_gate_running"} + {"requirement": "declaration-time experiment wall deadline and precise headless lead-repair attribution", "proof": "test_public_trials deadline regression and final affected learning/lifecycle run", "result": "behavioral_pass_in_71_test_gate_with_two_stale_schema_assertions"} ], "passed_gates": [ {"command": "python3 -m unittest test_objective_outcomes test_public_trials test_cli test_store test_service", "tests": 88, "seconds": 61.236, "resource_warnings": "promoted_to_errors", "offline_build_interpreter": "explicit_python3.12", "result": "pass_before_final_deadline_and_lead_attribution_refinement"}, @@ -23,7 +23,8 @@ "Headless receipt attempts do not include IDs on successful semantic records; keyed failure lookup was corrected (1 error in 7 tests/9.472s).", "Existing generic fake-runner native exit receipt lacks managed run/state fields; explicitly fixture-gated normalization now preserves that original receipt (1 error in 3 tests/15.093s).", "New whole-chain test used an invalid invented template mode human_reviewed; corrected to the existing reviewed contract, not a source policy relaxation (1 error in 6 tests/6.586s).", - "An exploratory fixture diagnostic requested a nonexistent exit_code column and failed before printing; the corrected diagnostic used finally cleanup. This is not a passing test or a production fault." + "An exploratory fixture diagnostic requested a nonexistent exit_code column and failed before printing; the corrected diagnostic used finally cleanup. This is not a passing test or a production fault.", + "The 71-test frozen affected learning/lifecycle gate completed in 233.881s with two stale schema-15 assertion failures (test_learning and test_lifecycle). No behavioral errors; the assertions were updated to schema 16, not removed." ], "open_gates": ["learning/lifecycle affected suite", "full core suite", "independent exact-patch review", "schema-16 safe upgrade and installed transport proof", "R6/R7/R8 separate acceptance"], "policy": {"automatic_experiments": "off", "automatic_promotion": "not_enabled", "new_provider_calls_for_this_slice": 0, "paid_api_fallback": false, "legacy_outcomes_rewritten": false}, diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index 398661c..e097cac 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -185,7 +185,7 @@ def test_migration_nine_creates_capacity_ledger(self): version = store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 15) + self.assertEqual(version, 16) tables = { row[0] for row in store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_decision_store.py b/test/core/test_decision_store.py index ab17148..b40704b 100644 --- a/test/core/test_decision_store.py +++ b/test/core/test_decision_store.py @@ -191,7 +191,7 @@ def test_schema_thirteen_contains_decision_cache_and_run_links(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 15) + self.assertEqual(version, 16) tables = { row[0] for row in self.store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_experiment_assignment_store.py b/test/core/test_experiment_assignment_store.py index 1e3c6a1..9946bd2 100644 --- a/test/core/test_experiment_assignment_store.py +++ b/test/core/test_experiment_assignment_store.py @@ -189,7 +189,7 @@ def test_schema_13_migration_preserves_legacy_evaluation_bytes(self): saved = upgraded.connection.execute('SELECT * FROM experiments').fetchone() self.assertEqual(saved['spec_json'], spec) self.assertEqual(saved['evaluation_json'], evaluation) - self.assertEqual(upgraded.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0], 15) + self.assertEqual(upgraded.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0], 16) self.assertEqual(upgraded.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 0) diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 2293bfd..0f0b622 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -705,7 +705,7 @@ def test_installed_wheel_applies_schema_four_to_twelve(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 15 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 16 assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_learning.py b/test/core/test_learning.py index aa480e7..7614c62 100644 --- a/test/core/test_learning.py +++ b/test/core/test_learning.py @@ -393,7 +393,7 @@ def test_current_schema_contains_outcome_ledger(self): store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0], - 15, + 16, ) columns = { row[1] for row in store.connection.execute("PRAGMA table_info(outcomes)") diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index 8ecc229..e5ceb66 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -552,7 +552,7 @@ def test_schema_twelve_contains_lifecycle_ledger(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 15) + self.assertEqual(version, 16) tables = { row[0] for row in self.store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", From f87060b7980a9fbe95314bb0986f84e424eccd3e Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 14:22:03 -0700 Subject: [PATCH 176/197] WIP checkpoint: R5 repair: transactional projection reads and active-worker deadline enforcement (2026-10-02 14:22) --- docs/plans/engineering-team/RESUME.md | 29 ++++++++++--- ...public-integration-partial-2026-10-02.json | 21 ++++++++- plugin/core/src/devsquad/attempt_runner.py | 3 +- .../core/src/devsquad/objective_outcomes.py | 5 ++- plugin/core/src/devsquad/store.py | 43 +++++++++++++++---- test/core/test_objective_outcomes.py | 18 ++++++++ test/core/test_public_trials.py | 23 ++++++++++ 7 files changed, 123 insertions(+), 19 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 85e2b5b..308588b 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -51,12 +51,29 @@ Public concurrent-budget and promotion → new-run binding → held-out regressi → rollback tests pass. See `evidence/R5-public-integration-partial-2026-10-02.json`. The frozen learning/lifecycle affected gate finished: **71 tests/233.881s**, -two stale schema-15 assertion failures, no behavioral errors. Those expectations -now match schema 16; keep the failed gate in partial evidence. No test process -remains active at this checkpoint. Next: recheck the two assertions, then the -full core gate with explicit offline Python 3.12 build interpreter, -independent exact-patch review, schema-16 safe installer/installed transport -recheck and R5 closure. Current production installation/ledger is still the +two stale schema-15 assertion failures, no behavioral errors. The two direct +assertion checks and another 22 migration/capacity/decision tests passed after +updating expectations, not removing them. Initial independent native R5 audit +**`5a70f3db-91a8-4606-9809-c0b993b988b1`** rejected exact `14b3a391`: +medium pending-projection nested read transaction and high reservation-only +experiment deadline. Diff/Bash/16 public tests/reference checks had passed; +that does not supersede the findings. Controlled active-worker reproduction +confirmed 7.274s work against a 3s experiment cap. The root full gate was +explicitly SIGINT-stopped (PID 77765/exit 130) to repair this known rejected +candidate, not counted as a pass. That interruption bypassed unittest cleanup, +producing TemporaryDirectory/SQLite finalizer warnings; normal complete gates +must still prove no warnings/unraisable. No scoped leftover process was found. + +Repairs now move project projection before the consistent proposal read, +clip launch timeouts to the experiment deadline and poll remaining budget in +the gated runner for active work. A generic fixture timeout's native TIMEOUT +field is preserved and projected, not rewritten. Three new red public tests +reproduced the defects (1 failure/2 errors/14.842s); **54 affected outcome, +trial, store, supervisor and M2 gate tests now pass in 48.572s**. No test/native +process remains active at this checkpoint. Next: full core gate with explicit +offline Python 3.12 build interpreter and narrow independent exact-repair +follow-up, then schema-16 safe installer/installed transport recheck and R5 +closure. Current production installation/ledger is still the accepted R4 **schema 15**; never open it with source Service while schema-16 work is incomplete. Preserve failed red/intermediate probes truthfully. Continue R6 UX/readiness/generated-reference checks, R7/C1 and final R8 diff --git a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json index 0a70205..c5acfa2 100644 --- a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json @@ -14,6 +14,7 @@ {"requirement": "declaration-time experiment wall deadline and precise headless lead-repair attribution", "proof": "test_public_trials deadline regression and final affected learning/lifecycle run", "result": "behavioral_pass_in_71_test_gate_with_two_stale_schema_assertions"} ], "passed_gates": [ + {"command": "python3 -m unittest test_objective_outcomes test_public_trials test_store test_supervisor test_m2_supervisor_gate", "tests": 54, "seconds": 48.572, "resource_warnings": "promoted_to_errors", "result": "pass_after_independent_findings_repair"}, {"command": "python3 -m unittest test_objective_outcomes test_public_trials test_cli test_store test_service", "tests": 88, "seconds": 61.236, "resource_warnings": "promoted_to_errors", "offline_build_interpreter": "explicit_python3.12", "result": "pass_before_final_deadline_and_lead_attribution_refinement"}, {"command": "python3 -m unittest test_public_trials.PublicTrialTest.test_public_outcomes_evaluate_qualify_promote_new_run_and_roll_back", "tests": 1, "seconds": 12.300, "result": "pass"} ], @@ -24,9 +25,25 @@ "Existing generic fake-runner native exit receipt lacks managed run/state fields; explicitly fixture-gated normalization now preserves that original receipt (1 error in 3 tests/15.093s).", "New whole-chain test used an invalid invented template mode human_reviewed; corrected to the existing reviewed contract, not a source policy relaxation (1 error in 6 tests/6.586s).", "An exploratory fixture diagnostic requested a nonexistent exit_code column and failed before printing; the corrected diagnostic used finally cleanup. This is not a passing test or a production fault.", - "The 71-test frozen affected learning/lifecycle gate completed in 233.881s with two stale schema-15 assertion failures (test_learning and test_lifecycle). No behavioral errors; the assertions were updated to schema 16, not removed." + "The 71-test frozen affected learning/lifecycle gate completed in 233.881s with two stale schema-15 assertion failures (test_learning and test_lifecycle). No behavioral errors; the assertions were updated to schema 16, not removed. The two direct checks and another 22 migration/capacity/decision tests (1.208s) then passed.", + "Independent native audit rejected exact 14b3a391 for pending-projection nested transaction and active-worker experiment deadline gaps, despite passing required diff/Bash/16 public tests/reference checks.", + "Controlled 3s experiment / 6s worker diagnostic reached resume_candidate_review after 7.274s, confirming the deadline gap. Temporary diagnostic run was cancelled and cleaned.", + "Root full gate on that rejected candidate was intentionally SIGINT-stopped at verified PID 77765, exit 130; not a timeout or pass. unittest interruption skipped cleanup and emitted TemporaryDirectory and SQLite finalizer ResourceWarnings; no scoped remaining process was found.", + "Three public red regressions reproduced proposal nested BEGIN, generic TIMEOUT receipt normalization and active-worker deadline overrun (1 failure/2 errors/14.842s). Fixed source passes the 54-test affected gate." ], + "rejected_independent_review": { + "run_id": "5a70f3db-91a8-4606-9809-c0b993b988b1", "state": "failed", "version": 22, "host_disposition": "reject", + "base_revision": "d3f7c0b3cc1c704dee32f96a87500a3284572d9b", "target_revision": "14b3a3910972484074ed52d99f70123944e56fd0", + "candidate_sha256": "2aa50bf50541ebabe337ce7d9f1df99ce7151c0a888f08a195efd6dc0239c6f9", + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "findings": [{"id": "R5-001", "severity": "medium", "title": "Pending projection breaks transactional policy proposal reads", "status": "source_repaired_focused_pass_followup_pending"}, {"id": "R5-002", "severity": "high", "title": "Experiment wall deadline only fences reservations", "status": "source_repaired_focused_pass_followup_pending"}], + "review_sha256": "912c79e25ebef6cd51ed00fb118aa761b80a6301f19937ca948298d570e52529", + "checks_sha256": "42204bc469b2298e66602e3864f70c8a9e7e70f9d4131da37b322c72c343c9ca", + "receipt_sha256": "c63483b4835fc5eb27afa30dd657436a9b5e595166d8ba765ead35ead934e3e5", + "checks": {"diff_bash_reference": "pass", "public_tests": 16, "seconds": 37.184, "integrity": "verified_unchanged"}, + "usage": {"source": "native_reported", "input_tokens": 196084, "output_tokens": 1316, "total_tokens": 197400} + }, "open_gates": ["learning/lifecycle affected suite", "full core suite", "independent exact-patch review", "schema-16 safe upgrade and installed transport proof", "R6/R7/R8 separate acceptance"], - "policy": {"automatic_experiments": "off", "automatic_promotion": "not_enabled", "new_provider_calls_for_this_slice": 0, "paid_api_fallback": false, "legacy_outcomes_rewritten": false}, + "policy": {"automatic_experiments": "off", "automatic_promotion": "not_enabled", "bounded_native_review_runs": 1, "paid_api_fallback": false, "legacy_outcomes_rewritten": false}, "limits": "Fixture outcomes demonstrate runtime/provenance behavior, not real model superiority or subscription cost savings. Installed runtime remains the accepted R4 release until acceptance gates pass." } diff --git a/plugin/core/src/devsquad/attempt_runner.py b/plugin/core/src/devsquad/attempt_runner.py index cb7b127..a4a3291 100644 --- a/plugin/core/src/devsquad/attempt_runner.py +++ b/plugin/core/src/devsquad/attempt_runner.py @@ -50,7 +50,8 @@ def main(argv=None): try: while child.poll() is None: run=store.run(a.run_id) - if run["state"]=="cancelling" or time.monotonic()>=deadline: + if (run["state"]=="cancelling" or time.monotonic()>=deadline + or store.remaining_wall_seconds(a.run_id) == 0): cancelled=run["state"]=="cancelling"; timed_out=not cancelled if inspect_process(child.pid,child.pid,started)!="live": raise RuntimeError("child identity became unsafe") try: os.killpg(child.pid,signal.SIGTERM) diff --git a/plugin/core/src/devsquad/objective_outcomes.py b/plugin/core/src/devsquad/objective_outcomes.py index a419fa3..c315730 100644 --- a/plugin/core/src/devsquad/objective_outcomes.py +++ b/plugin/core/src/devsquad/objective_outcomes.py @@ -11,13 +11,14 @@ def project_outcome(run: dict[str, Any], receipt: dict[str, Any], attempts: list[dict[str, Any]], artifact_refs: dict[str, str], assignment: dict[str, Any] | None) -> dict[str, Any]: snapshot = json.loads(run["mutable_snapshot"] or "null") - if ("internal_fake_delay" in (snapshot or {}) and set(receipt) == { + if ("internal_fake_delay" in (snapshot or {}) and set(receipt) - {"error"} == { "returncode", "cancelled", "timed_out", "stdout", "stderr", "finished_at"}): # The existing generic runner fixture retains its native exit receipt, # unlike the managed workflow's semantic receipt. Do not rewrite it. if (type(receipt["returncode"]) is not int or type(receipt["cancelled"]) is not bool or type(receipt["timed_out"]) is not bool - or type(receipt["finished_at"]) not in {int, float}): + or type(receipt["finished_at"]) not in {int, float} + or "error" in receipt and (receipt["error"] != "TIMEOUT" or not receipt["timed_out"])): raise ContractError("objective generic exit receipt is invalid") expected = "cancelled" if receipt["cancelled"] else "failed" if receipt["timed_out"] or receipt["returncode"] != 0 else "succeeded" if run["state"] != expected and run["state"] != "cancelled": diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 716c5a5..bd065f3 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -799,11 +799,26 @@ def remaining_wall_seconds( if run is None: raise ContractError("run does not exist") wall_seconds = self._wall_seconds_from_run(run) + current = _authoritative_now(now) + experiment_remaining = self._experiment_remaining_wall_seconds(run_id, current) if wall_seconds is None: - return None + return experiment_remaining elapsed_ms = self._execution_elapsed_ms( - run_id, run, _authoritative_now(now), + run_id, run, current, ) + remaining = max(0, (wall_seconds * 1000 - elapsed_ms) // 1000) + return remaining if experiment_remaining is None else min(remaining, experiment_remaining) + + def _experiment_remaining_wall_seconds(self, run_id: str, current: datetime) -> int | None: + row = self.connection.execute( + "SELECT s.spec_json,s.recorded_at FROM experiment_assignments a " + "JOIN experiment_specs s ON s.experiment_id=a.experiment_id WHERE a.run_id=?", + (run_id,), + ).fetchone() + if row is None: + return None + wall_seconds = json.loads(row["spec_json"])["budget"]["wall_seconds"] + elapsed_ms = max(0, int((current - _parse_utc(row["recorded_at"])).total_seconds() * 1000)) return max(0, (wall_seconds * 1000 - elapsed_ms) // 1000) def _enforce_attempt_budget( @@ -1795,12 +1810,26 @@ def outcomes_for_run(self, run_id: str) -> list[dict[str, Any]]: ) ] + def repair_project_outcomes(self, project: Path) -> None: + """Repair before, never inside, a caller's consistent read transaction.""" + if self.connection.in_transaction: + raise ConflictError("project outcome repair requires an independent transaction") + common = str(git_common_dir(project)) + for job in self.connection.execute( + "SELECT j.run_id FROM objective_outcome_jobs j JOIN runs r ON r.id=j.run_id " + "JOIN projects p ON p.id=r.project_id WHERE p.git_common_dir=? " + "AND j.completed_outcome_id IS NULL AND r.state IN ('succeeded','failed','cancelled')", + (common,)).fetchall(): + self.project_final_outcome(job["run_id"]) + def learning_report( - self, project: Path, *, now: datetime | None = None, + self, project: Path, *, now: datetime | None = None, _repair_pending: bool = True, ) -> dict[str, Any]: """Repair pending public projections and compare with explicit missingness.""" from .learning import build_comparison_report + if _repair_pending: + self.repair_project_outcomes(project) common_dir = git_common_dir(project) project_row = self.connection.execute( "SELECT id FROM projects WHERE git_common_dir=?", (str(common_dir),), @@ -1820,8 +1849,6 @@ def learning_report( (project_id,), ) ] - for terminal_run in terminal_runs: - self.project_final_outcome(terminal_run["run_id"]) outcome_records = [ {"run_id": row["run_id"], "outcome": json.loads(row["payload_json"])} for row in self.connection.execute( @@ -2004,9 +2031,10 @@ def learning_proposal_inputs( from .experiment_eligibility import current_evidence, saved_evaluation project_path = project.resolve(strict=True) current = _authoritative_now(now) + self.repair_project_outcomes(project_path) self.connection.execute("BEGIN") try: - report = self.learning_report(project_path, now=current) + report = self.learning_report(project_path, now=current, _repair_pending=False) if report["project_id"] is None: row = self.connection.execute( "SELECT experiment_id FROM experiments WHERE project_path=? " @@ -2963,8 +2991,7 @@ def _enforce_experiment_budget(self, run_id: str) -> None: ).fetchone()[0] if consumed >= spec["budget"]["max_worker_invocations"]: raise BudgetExhausted("experiment worker reservation budget is exhausted") - elapsed = (_authoritative_now() - datetime.fromisoformat(row["recorded_at"])).total_seconds() - if elapsed >= spec["budget"]["wall_seconds"]: + if self._experiment_remaining_wall_seconds(run_id, _authoritative_now()) == 0: raise BudgetExhausted("experiment wall-time budget is exhausted") def reserve_attempt( diff --git a/test/core/test_objective_outcomes.py b/test/core/test_objective_outcomes.py index 799d131..021bf94 100644 --- a/test/core/test_objective_outcomes.py +++ b/test/core/test_objective_outcomes.py @@ -138,6 +138,24 @@ def test_corrupt_pending_projection_does_not_block_other_runs(self): self.assertEqual(self.service.status(good_id)["state"], "succeeded") self.assertEqual(self.outcome(good_id)["verdict"], "succeeded") + def test_proposal_repairs_pending_outcome_before_its_consistent_read(self): + run_id, waiting = self.fixture.start_waiting("objective-proposal-crash") + with mock.patch.object(Store, "project_final_outcome", side_effect=RuntimeError("projection crash")): + with self.assertRaises(RuntimeError): + self.accept(run_id, waiting) + proposed = self.service.learning_propose(self.fixture.repo) + self.assertEqual(proposed["proposal"]["sample_sizes"]["final_outcomes"], 1) + self.assertEqual(self.outcome(run_id)["verdict"], "succeeded") + + def test_generic_worker_timeout_preserves_native_exit_and_projects_failure(self): + task = copy.deepcopy(self.fixture.task) + task["budget"]["wall_seconds"] = 2 + started = self.service.start(task, "objective-generic-timeout", _internal_fake_delay=4) + self.fixture.wait_state(started["run_id"], {"failed"}) + final = self.outcome(started["run_id"]) + self.assertEqual(final["verdict"], "failed") + self.assertEqual([c["result"] for c in final["contributions"]], ["failed"]) + if __name__ == "__main__": unittest.main() diff --git a/test/core/test_public_trials.py b/test/core/test_public_trials.py index ef4c4bd..0f5c5b4 100644 --- a/test/core/test_public_trials.py +++ b/test/core/test_public_trials.py @@ -119,6 +119,29 @@ def test_experiment_deadline_blocks_later_launch_without_invented_exposure(self) finally: store.close() + def test_experiment_deadline_also_stops_an_already_running_worker(self): + fixture = ExperimentRuntimeFixture(self.fixture.root / "active-deadline", workflow="issue-delivery") + self.addCleanup(fixture.close) + fixture.spec["budget"]["wall_seconds"] = 5 + before = time.monotonic() + started = fixture.service.trial_start( + fixture.spec, "eval-1", "candidate", fixture.task("eval-1", "candidate"), "public-active-deadline", + _internal_implementation_fixture={"writes": [{"path": "README", "content": "fixed eval-1\n"}], "delay_seconds": 10}, + _internal_review_fixture={"verdict": "clean", "summary": "Active deadline fixture.", "findings": []}, + ) + fixture.runs[("eval-1", "candidate")] = started["run_id"] + self.assertEqual(fixture.wait(started["run_id"])["state"], "failed") + self.assertLess(time.monotonic() - before, 8, "worker must not run for its separate 120-second task budget") + store = fixture.store() + try: + attempts = store.attempts_for_run(started["run_id"]) + self.assertEqual(len(attempts), 1) + self.assertIsNotNone(attempts[0]["pid"], "this must exercise active work, not a prelaunch failure") + final = store.outcomes_for_run(started["run_id"])[0]["outcome"] + self.assertEqual([c["result"] for c in final["contributions"]], ["failed"]) + finally: + store.close() + def test_public_controller_rejects_delivery_reviewer_and_unbounded_requests(self): task = self.fixture.task("eval-1", "control") task.update(workflow="issue-delivery") From bf3de0867484552d354e6e8b6ba835f31a93aeb3 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 14:31:25 -0700 Subject: [PATCH 177/197] WIP checkpoint: test: wait for terminal headless integrity receipts and checkpoint R5 repair acceptance (2026-10-02 14:31) --- docs/plans/engineering-team/RESUME.md | 15 +++++++++---- docs/plans/engineering-team/backlog.json | 2 +- ...public-integration-partial-2026-10-02.json | 21 ++++++++++++++++--- test/core/test_check_integrity_runtime.py | 8 ++++++- 4 files changed, 37 insertions(+), 9 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 308588b..27a861c 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -69,10 +69,17 @@ clip launch timeouts to the experiment deadline and poll remaining budget in the gated runner for active work. A generic fixture timeout's native TIMEOUT field is preserved and projected, not rewritten. Three new red public tests reproduced the defects (1 failure/2 errors/14.842s); **54 affected outcome, -trial, store, supervisor and M2 gate tests now pass in 48.572s**. No test/native -process remains active at this checkpoint. Next: full core gate with explicit -offline Python 3.12 build interpreter and narrow independent exact-repair -follow-up, then schema-16 safe installer/installed transport recheck and R5 +trial, store, supervisor and M2 gate tests now pass in 48.572s**. Exact native +repair follow-up **e046a174-55ce-4496-973e-2fc97c68a493** is clean and accepted, +succeeded/22; diff/Bash/13 public tests/reference checks passed with unchanged +integrity. The next full fail-fast gate stopped at 17 tests/10.494s, one failure, +no errors/skips/unraisable: a headless integrity test requested a terminal +receipt during a transient handoff transition. Isolated reproduction passed; +the test now waits for terminal state in headless mode while retaining every +integrity assertion. **66 integrity/outcome/review/delivery tests pass in +125.322s**. No test/native process remains active at this checkpoint. Next: +one full core gate with explicit offline Python 3.12 build interpreter, then +schema-16 safe installer/installed transport recheck and R5 closure. Current production installation/ledger is still the accepted R4 **schema 15**; never open it with source Service while schema-16 work is incomplete. Preserve failed red/intermediate probes truthfully. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 45b8cae..9ec5740 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -46,7 +46,7 @@ {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, - {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "in_progress", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-public-integration-partial-2026-10-02.json", "checkpoint": "Schema-16 objective projection and explicit public trial/controller connections implemented, not yet accepted or installed. 88 affected tests pass; public concurrent-budget and promotion/new-run/held-out-regression/rollback chain pass. Frozen learning/lifecycle gate running; full integration, independent review and installed upgrade/transport proof remain open. No automatic experimentation or provider-quality claim."}, + {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "in_progress", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-public-integration-partial-2026-10-02.json", "checkpoint": "Schema-16 objective projection and explicit public trial/controller connections implemented, not yet accepted or installed. Initial native audit rejected two defects; repaired exact follow-up e046a174 is clean/accepted with all mandatory checks passing. 54 repaired-path and 66 integrity/outcome/review/delivery tests pass. Failed/interrupted full gates retained truthfully; next full gate and installed upgrade/transport proof remain open. No automatic experimentation or provider-quality claim."}, {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"G4 discovery verified by accepted native 477-test delivery and combined review; integrated and installed at 6d2e0ba. Remaining UX and R4/R5 dependencies stay open."}, {"id": "R7", "title": "Complete Council within existing runner", "status": "pending", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"]}, {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} diff --git a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json index c5acfa2..5685b04 100644 --- a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json @@ -14,6 +14,7 @@ {"requirement": "declaration-time experiment wall deadline and precise headless lead-repair attribution", "proof": "test_public_trials deadline regression and final affected learning/lifecycle run", "result": "behavioral_pass_in_71_test_gate_with_two_stale_schema_assertions"} ], "passed_gates": [ + {"command": "python3 -m unittest test_check_integrity_runtime test_objective_outcomes test_review_runtime test_delivery_workflow", "tests": 66, "seconds": 125.322, "resource_warnings": "promoted_to_errors", "result": "pass_after_headless_test_transition_fix"}, {"command": "python3 -m unittest test_objective_outcomes test_public_trials test_store test_supervisor test_m2_supervisor_gate", "tests": 54, "seconds": 48.572, "resource_warnings": "promoted_to_errors", "result": "pass_after_independent_findings_repair"}, {"command": "python3 -m unittest test_objective_outcomes test_public_trials test_cli test_store test_service", "tests": 88, "seconds": 61.236, "resource_warnings": "promoted_to_errors", "offline_build_interpreter": "explicit_python3.12", "result": "pass_before_final_deadline_and_lead_attribution_refinement"}, {"command": "python3 -m unittest test_public_trials.PublicTrialTest.test_public_outcomes_evaluate_qualify_promote_new_run_and_roll_back", "tests": 1, "seconds": 12.300, "result": "pass"} @@ -29,7 +30,8 @@ "Independent native audit rejected exact 14b3a391 for pending-projection nested transaction and active-worker experiment deadline gaps, despite passing required diff/Bash/16 public tests/reference checks.", "Controlled 3s experiment / 6s worker diagnostic reached resume_candidate_review after 7.274s, confirming the deadline gap. Temporary diagnostic run was cancelled and cleaned.", "Root full gate on that rejected candidate was intentionally SIGINT-stopped at verified PID 77765, exit 130; not a timeout or pass. unittest interruption skipped cleanup and emitted TemporaryDirectory and SQLite finalizer ResourceWarnings; no scoped remaining process was found.", - "Three public red regressions reproduced proposal nested BEGIN, generic TIMEOUT receipt normalization and active-worker deadline overrun (1 failure/2 errors/14.842s). Fixed source passes the 54-test affected gate." + "Three public red regressions reproduced proposal nested BEGIN, generic TIMEOUT receipt normalization and active-worker deadline overrun (1 failure/2 errors/14.842s). Fixed source passes the 54-test affected gate.", + "The next complete fail-fast core invocation stopped after 17 tests/10.494s (one failure, no errors/skips/unraisable): the integrity test inspected a terminal receipt during the transient headless awaiting_host-to-queued lead transition. The isolated test passed in 2.719s; the test now waits for terminal state only in headless mode, retaining every integrity/source-preservation assertion. All 66 integrity/outcome/review/delivery tests pass in 125.322s. This is not a quota timeout or a passing full gate." ], "rejected_independent_review": { "run_id": "5a70f3db-91a8-4606-9809-c0b993b988b1", "state": "failed", "version": 22, "host_disposition": "reject", @@ -43,7 +45,20 @@ "checks": {"diff_bash_reference": "pass", "public_tests": 16, "seconds": 37.184, "integrity": "verified_unchanged"}, "usage": {"source": "native_reported", "input_tokens": 196084, "output_tokens": 1316, "total_tokens": 197400} }, - "open_gates": ["learning/lifecycle affected suite", "full core suite", "independent exact-patch review", "schema-16 safe upgrade and installed transport proof", "R6/R7/R8 separate acceptance"], - "policy": {"automatic_experiments": "off", "automatic_promotion": "not_enabled", "bounded_native_review_runs": 1, "paid_api_fallback": false, "legacy_outcomes_rewritten": false}, + "accepted_independent_repair_review": { + "run_id": "e046a174-55ce-4496-973e-2fc97c68a493", "state": "succeeded", "version": 22, "host_disposition": "accept", + "base_revision": "14b3a3910972484074ed52d99f70123944e56fd0", "target_revision": "f87060b7980a9fbe95314bb0986f84e424eccd3e", + "candidate_sha256": "564e2e937dc01d1127378537b45b7a9170b91c8a90545e7d3775321025640aa2", + "verdict": "clean", "findings": [], "closed_findings": ["R5-001", "R5-002"], + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "review_sha256": "a448ab152287e163ac4610850bee574939f97fb822863d31e8d8050fdf09ac6f", + "checks_sha256": "a1a74d34c4fda3960d880ea018fbf30f67e38f92d026cfe63f40e56b207c86de", + "receipt_sha256": "58eea09f033104b98b0dd5ef086ecbaf65faa050802d39bb4da4bf08d8c5b254", + "checks": {"diff_bash_reference": "pass", "public_tests": 13, "seconds": 10.729, "integrity": "verified_unchanged"}, + "usage": {"source": "native_reported", "input_tokens": 131692, "output_tokens": 1078, "total_tokens": 132770}, + "scope": "Only the declared repair diff is accepted; whole R5 still requires the full and installation gates. Subsequent change is test-only headless observation timing." + }, + "open_gates": ["full core suite", "schema-16 safe upgrade and installed transport proof", "R6/R7/R8 separate acceptance"], + "policy": {"automatic_experiments": "off", "automatic_promotion": "not_enabled", "bounded_native_review_runs": 2, "paid_api_fallback": false, "legacy_outcomes_rewritten": false}, "limits": "Fixture outcomes demonstrate runtime/provenance behavior, not real model superiority or subscription cost savings. Installed runtime remains the accepted R4 release until acceptance gates pass." } diff --git a/test/core/test_check_integrity_runtime.py b/test/core/test_check_integrity_runtime.py index 486843b..dd46409 100644 --- a/test/core/test_check_integrity_runtime.py +++ b/test/core/test_check_integrity_runtime.py @@ -208,7 +208,13 @@ def exercise_mutation(self, workflow, mode, *, add_source=False): ), ] run_id = self.start(workflow, mode, checks) - status = self.wait_state(run_id, {"awaiting_host", "succeeded", "failed"}) + # Headless mode traverses awaiting_host while it queues its own lead. + # Only host mode may inspect that intermediate packet; headless evidence + # is the terminal receipt, which is not published at the transition. + states = {"succeeded", "failed"} + if mode == "host": + states.add("awaiting_host") + status = self.wait_state(run_id, states) if mode == "host": self.assertEqual(status["state"], "awaiting_host", status) claimed = self.service.handoff_claim(run_id, status["version"], "integrity-host") From dfe9976224e9ff2166a016b802e074aa1f763c4b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 14:37:41 -0700 Subject: [PATCH 178/197] WIP checkpoint: test: retain schema-four upgrade assertion for schema 16 (2026-10-02 14:37) --- docs/plans/engineering-team/RESUME.md | 16 +++++++++++++++- ...R5-public-integration-partial-2026-10-02.json | 3 ++- test/core/test_handoff_store.py | 2 +- 3 files changed, 18 insertions(+), 3 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 27a861c..8801011 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -77,7 +77,12 @@ no errors/skips/unraisable: a headless integrity test requested a terminal receipt during a transient handoff transition. Isolated reproduction passed; the test now waits for terminal state in headless mode while retaining every integrity assertion. **66 integrity/outcome/review/delivery tests pass in -125.322s**. No test/native process remains active at this checkpoint. Next: +125.322s**. Frozen bf3de086 fail-fast full invocation stopped at **188 tests / +274.410s**, one remaining stale schema-four upgrade assertion (15 versus actual +16); zero errors/skips/unraisable, UTC/monotonic ~274.50s. Only its expected +current schema changed; the direct migration regression passes in 0.090s and +other migration assertions were audited without changing historical versions. +No test/native process remains active at this checkpoint. Next: one full core gate with explicit offline Python 3.12 build interpreter, then schema-16 safe installer/installed transport recheck and R5 closure. Current production installation/ledger is still the @@ -86,6 +91,15 @@ work is incomplete. Preserve failed red/intermediate probes truthfully. Continue R6 UX/readiness/generated-reference checks, R7/C1 and final R8 acceptance after R5; whole-plan completion is not claimed. +R6 terminal and doctor slices are delegated in isolated attached worktrees +`r6-terminal` and `r6-readiness` at bf3de086. R7 Council is implementing in +isolated `r7-council` from the same baseline. They must not alter this frozen +candidate or production installation. Integrate their checkpoint commits +only after this R5 gate; preserve their work if interrupted. macOS default- +deny sandbox probe establishes own-evidence access with peer/ledger denial, +not native Council compatibility or acceptance. No new native Council call +has been made. + Workspace: /Users/Dikshant/Desktop/Projects/devsquad. Branch: `codex/engineering-team`; never restart this build from main. Source runtime repairs and the accepted G4 candidate below are **integrated diff --git a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json index 5685b04..38d0b44 100644 --- a/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R5-public-integration-partial-2026-10-02.json @@ -31,7 +31,8 @@ "Controlled 3s experiment / 6s worker diagnostic reached resume_candidate_review after 7.274s, confirming the deadline gap. Temporary diagnostic run was cancelled and cleaned.", "Root full gate on that rejected candidate was intentionally SIGINT-stopped at verified PID 77765, exit 130; not a timeout or pass. unittest interruption skipped cleanup and emitted TemporaryDirectory and SQLite finalizer ResourceWarnings; no scoped remaining process was found.", "Three public red regressions reproduced proposal nested BEGIN, generic TIMEOUT receipt normalization and active-worker deadline overrun (1 failure/2 errors/14.842s). Fixed source passes the 54-test affected gate.", - "The next complete fail-fast core invocation stopped after 17 tests/10.494s (one failure, no errors/skips/unraisable): the integrity test inspected a terminal receipt during the transient headless awaiting_host-to-queued lead transition. The isolated test passed in 2.719s; the test now waits for terminal state only in headless mode, retaining every integrity/source-preservation assertion. All 66 integrity/outcome/review/delivery tests pass in 125.322s. This is not a quota timeout or a passing full gate." + "The next complete fail-fast core invocation stopped after 17 tests/10.494s (one failure, no errors/skips/unraisable): the integrity test inspected a terminal receipt during the transient headless awaiting_host-to-queued lead transition. The isolated test passed in 2.719s; the test now waits for terminal state only in headless mode, retaining every integrity/source-preservation assertion. All 66 integrity/outcome/review/delivery tests pass in 125.322s. This is not a quota timeout or a passing full gate.", + "Frozen bf3de086 fail-fast gate stopped after 188 tests/274.410s, one stale test_handoff_store schema-four upgrade expectation (15 vs actual 16), no errors/skips/unraisable; UTC/monotonic both 274.50s. Corrected only the current-schema expected value; all schema_migrations expectations were searched to distinguish preserved historical schemas. Direct migration regression passes in 0.090s. This is not a runtime migration defect, quota timeout or passing full gate." ], "rejected_independent_review": { "run_id": "5a70f3db-91a8-4606-9809-c0b993b988b1", "state": "failed", "version": 22, "host_disposition": "reject", diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 0f0b622..fcd6819 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -197,7 +197,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 15, + 16, ) tables = { row[0] From 8f2c1c6a7d9edfe7cc3d78ecbc59ce105be636b2 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 14:53:43 -0700 Subject: [PATCH 179/197] WIP checkpoint: docs: close accepted installed R5 outcomes and public trials gates (2026-10-02 14:53) --- docs/plans/engineering-team/RESUME.md | 76 +++++++------------ docs/plans/engineering-team/backlog.json | 14 ++-- .../evidence/R5-closure-2026-10-02.json | 41 ++++++++++ 3 files changed, 77 insertions(+), 54 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R5-closure-2026-10-02.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 8801011..03db090 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -36,7 +36,8 @@ Antigravity externally updated to **1.2.14**: current doctor correctly labels its adapter unverified; the old 1.2.13 live status receipt is not a new-version proof. Recheck this during R6/R8 without broad MCP listings or UI bypass. -**R5 is source-in-progress**, not accepted/installed. The new schema-16 +**R5 is now accepted and installed** at source **dfe9976**, schema **16**. +Closure matrix: `evidence/R5-closure-2026-10-02.json`. The new projection outbox opts in new public runs without rewriting legacy history. Terminal transitions and targeted status/result/report/evaluation replay generate one final outcome; failures, repairs, findings, fenced lead @@ -45,57 +46,38 @@ does not block unrelated runs. The explicit public `trial` / `trial_start` path now uses the schema-14 assignment fence; the fixture no longer patches preparation or manually imports finals. Experiment reservations share a durable all-attempt call cap and a declaration-time wall deadline; automatic -experimentation stays off. 88 affected outcome/service/CLI/store tests passed -in 61.236s (before the added deadline test/lead-repair attribution refinement). +experimentation stays off. Public concurrent-budget and promotion → new-run binding → held-out regression → rollback tests pass. See `evidence/R5-public-integration-partial-2026-10-02.json`. -The frozen learning/lifecycle affected gate finished: **71 tests/233.881s**, -two stale schema-15 assertion failures, no behavioral errors. The two direct -assertion checks and another 22 migration/capacity/decision tests passed after -updating expectations, not removing them. Initial independent native R5 audit -**`5a70f3db-91a8-4606-9809-c0b993b988b1`** rejected exact `14b3a391`: -medium pending-projection nested read transaction and high reservation-only -experiment deadline. Diff/Bash/16 public tests/reference checks had passed; -that does not supersede the findings. Controlled active-worker reproduction -confirmed 7.274s work against a 3s experiment cap. The root full gate was -explicitly SIGINT-stopped (PID 77765/exit 130) to repair this known rejected -candidate, not counted as a pass. That interruption bypassed unittest cleanup, -producing TemporaryDirectory/SQLite finalizer warnings; normal complete gates -must still prove no warnings/unraisable. No scoped leftover process was found. - -Repairs now move project projection before the consistent proposal read, -clip launch timeouts to the experiment deadline and poll remaining budget in -the gated runner for active work. A generic fixture timeout's native TIMEOUT -field is preserved and projected, not rewritten. Three new red public tests -reproduced the defects (1 failure/2 errors/14.842s); **54 affected outcome, -trial, store, supervisor and M2 gate tests now pass in 48.572s**. Exact native -repair follow-up **e046a174-55ce-4496-973e-2fc97c68a493** is clean and accepted, -succeeded/22; diff/Bash/13 public tests/reference checks passed with unchanged -integrity. The next full fail-fast gate stopped at 17 tests/10.494s, one failure, -no errors/skips/unraisable: a headless integrity test requested a terminal -receipt during a transient handoff transition. Isolated reproduction passed; -the test now waits for terminal state in headless mode while retaining every -integrity assertion. **66 integrity/outcome/review/delivery tests pass in -125.322s**. Frozen bf3de086 fail-fast full invocation stopped at **188 tests / -274.410s**, one remaining stale schema-four upgrade assertion (15 versus actual -16); zero errors/skips/unraisable, UTC/monotonic ~274.50s. Only its expected -current schema changed; the direct migration regression passes in 0.090s and -other migration assertions were audited without changing historical versions. -No test/native process remains active at this checkpoint. Next: -one full core gate with explicit offline Python 3.12 build interpreter, then -schema-16 safe installer/installed transport recheck and R5 -closure. Current production installation/ledger is still the -accepted R4 **schema 15**; never open it with source Service while schema-16 -work is incomplete. Preserve failed red/intermediate probes truthfully. -Continue R6 UX/readiness/generated-reference checks, R7/C1 and final R8 -acceptance after R5; whole-plan completion is not claimed. +Initial native R5 audit rejected two real defects. Project projection now +precedes the consistent proposal read; launch/active-worker deadlines include +the shared experiment deadline. Exact repair follow-up +**e046a174-55ce-4496-973e-2fc97c68a493** is clean/accepted, succeeded/22; +diff/Bash/13 public tests/reference checks pass with unchanged integrity. +The 54 repaired-path and 66 integrity/outcome/review/delivery tests pass. +Final frozen **504-test full gate passes in 511.241s**, two optional SDK skips, +zero errors/failures/unraisable, UTC/monotonic ~511.32s. Later revisions change +only test observations/schema expectations and documents, not reviewed code. + +Safe offline installation selected `0.1.0-py31214-90f1e87fb9ac-mcp-a26bc88afbef`; +zero active production runs, lazy migration to 16, zero retroactive objective +jobs. Reinstall no-op, payload drift false, pip check pass, four registrations +matching. **Nine actual installed SDK tests pass in 2.674s without skips.** +Installed normal review dry-run still resolves gpt-6.1-sol/low bounded trial, +without creating a run. Full installer tests include actual old-schema active +deferral, cancellation/recovery and migration to current16. Failed red tests, +rejected audit and interrupted/failed full gates remain in partial evidence; +do not repeat accepted proofs. No test/native process is active here. +Next: integrate/test R6 terminal/readiness, R7/C1 and final R8 acceptance; +whole-plan completion is not claimed. R6 terminal and doctor slices are delegated in isolated attached worktrees `r6-terminal` and `r6-readiness` at bf3de086. R7 Council is implementing in isolated `r7-council` from the same baseline. They must not alter this frozen -candidate or production installation. Integrate their checkpoint commits -only after this R5 gate; preserve their work if interrupted. macOS default- +candidate or production installation. Readiness checkpoint `dd709804` needs +a root-review follow-up for a SIGTERM-ignoring descendant after parent exit. +Integrate only clean checkpoint commits; preserve work if interrupted. macOS default- deny sandbox probe establishes own-evidence access with peer/ledger denial, not native Council compatibility or acceptance. No new native Council call has been made. @@ -151,8 +133,8 @@ use git-safety checkpoints, never stash. No goal is currently active. ## Current local installation and host proof Stable launcher: /Users/Dikshant/.local/bin/squad. -Selected release: `0.1.0-py31214-3631f1737bc9-mcp-a26bc88afbef`. -Python 3.12.14/MCP 2.2.0, schema 15; previous releases and private pre-upgrade +Selected release: `0.1.0-py31214-90f1e87fb9ac-mcp-a26bc88afbef`. +Python 3.12.14/MCP 2.2.0, schema 16; previous releases and private pre-upgrade SQLite backup retained. Source/plugin/installed payload drift is false. Upgrade defers for old active/recoverable runs, swaps without migration under the lock, then lazily migrates with old-client write guards. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 9ec5740..bd56022 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -14,14 +14,14 @@ "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", "next_milestone": "M5", - "next_work_package": "R5", + "next_work_package": "R6", "partial_implementation_checkpoint": { - "recorded_on": "2026-10-01", - "baseline_revision": "6d2e0ba", + "recorded_on": "2026-10-02", + "baseline_revision": "dfe9976", "status": "partial", - "artifact": "evidence/R8-installed-workflows-2026-10-01.json", - "scope": "Installed schema-15 runtime through 6d2e0ba; accepted real Claude-to-independent-Codex 477-test delivery, complete combined review and integration, actual Claude handoff/Grok operation and final Gemini CLI/MCP recheck", - "next_action": "R3 is accepted by evidence/R3-closure-2026-10-02.json. Finish R4's repaired account-wide pool fence full gate, narrow independent follow-up and installed recheck, then R5 public outcomes/trials, R6 UX and R7/C1. Preserve accepted earlier runtime proofs.", + "artifact": "evidence/R5-closure-2026-10-02.json", + "scope": "R3/R4/R5 accepted; installed schema16 objective outcomes/public trials with 504-test full gate, exact native repair review, safe old-schema upgrade tests and nine installed SDK tests. Earlier accepted Claude delivery/handoff and Grok/Gemini receipts retained separately.", + "next_action": "Integrate and verify delegated R6 terminal/readiness changes, then R7/C1 Council and final R8 acceptance. Do not repeat unchanged accepted proofs.", "limitations": "Whole-plan closure is not claimed. Required desktop UI proofs remain unverified; Antigravity IDE permission is denied. Grok/Gemini MCP status does not prove automatic writer/reviewer roles. Jev remains off with its one-request allowance spent; no paid API fallback or reset use authorized." }, "planning_checkpoint": { @@ -46,7 +46,7 @@ {"id": "R2", "title": "Observed Claude execution identity", "status": "complete", "milestones": ["M5"], "depends_on": [], "items": ["F2"], "evidence": "evidence/R2-observed-identity-2026-10-01.json", "checkpoint": "Source/offline repair verified at ef98889: final 367-test gate OK with two optional-SDK skips and stable UTC/monotonic timing; 227 Bash assertions and generated reference passed. Two independent-review findings repaired and independently rechecked. Earlier six-failure gate retained in evidence. SQLite warning remains R5; installed/live proof remains R8, not full M5 closure."}, {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, - {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "in_progress", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-public-integration-partial-2026-10-02.json", "checkpoint": "Schema-16 objective projection and explicit public trial/controller connections implemented, not yet accepted or installed. Initial native audit rejected two defects; repaired exact follow-up e046a174 is clean/accepted with all mandatory checks passing. 54 repaired-path and 66 integrity/outcome/review/delivery tests pass. Failed/interrupted full gates retained truthfully; next full gate and installed upgrade/transport proof remain open. No automatic experimentation or provider-quality claim."}, + {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "complete", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-closure-2026-10-02.json", "checkpoint": "Accepted dfe9976/source equivalent f87060b and installed schema16. Initial native findings repaired; exact follow-up e046a174 clean/accepted. Full 504 tests/511.241s, two optional SDK skips/no errors/failures/unraisable; nine actual installed SDK tests pass/no skips. Public complete trial/evaluate/qualify/promote/new-run/held-out regression/rollback, shared call/active deadline and all terminal projection origins pass. Safe old-schema upgrade and idempotence/drift gates pass. Failed history retained; automatic experimentation and Jev stay off."}, {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"G4 discovery verified by accepted native 477-test delivery and combined review; integrated and installed at 6d2e0ba. Remaining UX and R4/R5 dependencies stay open."}, {"id": "R7", "title": "Complete Council within existing runner", "status": "pending", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"]}, {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} diff --git a/docs/plans/engineering-team/evidence/R5-closure-2026-10-02.json b/docs/plans/engineering-team/evidence/R5-closure-2026-10-02.json new file mode 100644 index 0000000..927cf66 --- /dev/null +++ b/docs/plans/engineering-team/evidence/R5-closure-2026-10-02.json @@ -0,0 +1,41 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "accepted_source_and_installed_public_outcomes_trials", + "source_revision": "dfe9976224e9ff2166a016b802e074aa1f763c4b", + "reviewed_source_revision": "f87060b7980a9fbe95314bb0986f84e424eccd3e", + "source_scope_equivalence": "git diff --exit-code f87060b dfe997 -- plugin/core passed; later changes only test expectations/observation and checkpoint documents", + "full_gate": {"command": "env DEVSQUAD_BUILD_PYTHON= PYTHONPATH=plugin/core/src:test/core PYTHONWARNINGS=error::ResourceWarning python3 scripts/run-core-tests.py --failfast", "tests": 504, "seconds": 511.241, "failures": 0, "errors": 0, "skips": 2, "unraisable": [], "monotonic_seconds": 511.3239737499989, "utc_seconds": 511.325922, "source_and_tests_frozen": true}, + "focused_gates": [{"tests": 54, "seconds": 48.572, "scope": "review repairs/outcomes/trials/store/process ownership", "result": "pass"}, {"tests": 66, "seconds": 125.322, "scope": "integrity/outcomes/review/delivery", "result": "pass"}], + "independent_review": { + "initial_run": "5a70f3db-91a8-4606-9809-c0b993b988b1", "initial_disposition": "reject", "findings": ["R5-001", "R5-002"], + "repair_run": "e046a174-55ce-4496-973e-2fc97c68a493", "state": "succeeded", "version": 22, "disposition": "accept", "verdict": "clean", "remaining_findings": [], + "identity": {"harness": "codex", "harness_version": "codex-cli 0.159.2", "model_id": "gpt-6.1-sol", "effort": "low", "verification": "verified", "permission_policy": "read_only"}, + "candidate_sha256": "564e2e937dc01d1127378537b45b7a9170b91c8a90545e7d3775321025640aa2", + "review_sha256": "a448ab152287e163ac4610850bee574939f97fb822863d31e8d8050fdf09ac6f", + "checks_sha256": "a1a74d34c4fda3960d880ea018fbf30f67e38f92d026cfe63f40e56b207c86de", + "receipt_sha256": "58eea09f033104b98b0dd5ef086ecbaf65faa050802d39bb4da4bf08d8c5b254", + "checks": {"diff_bash_reference": "pass", "public_tests": 13, "seconds": 10.729, "integrity": "verified_unchanged"} + }, + "requirements": [ + {"requirement": "one objective final per new public run from every terminal origin; replay after terminal/projection crash", "proof": "test_objective_outcomes public preparation failure/cancel, worker failure, host/headless completion, timeout and replay tests", "result": "pass"}, + {"requirement": "failed/fallback/revision/finding/lead-repair attribution; no invented independent success or criterion-level test proof", "proof": "test_objective_outcomes, learning/lifecycle and full gate", "result": "pass"}, + {"requirement": "append-only late corrections; legacy history unchanged; scoped pending projection repair and consistent proposal read", "proof": "public report/proposal replay, manual final replacement rejection, corruption isolation and native R5-001 follow-up", "result": "pass"}, + {"requirement": "public bounded trials freeze predeclared spec/assignment before real attempts, with one shared all-attempt cap", "proof": "test_public_trials concurrency/fallback/declaration replay; fixture now calls trial_start without preparation monkeypatch or manual final import", "result": "pass"}, + {"requirement": "experiment wall deadline stops already launched work with owned-child cleanup", "proof": "public actual delayed worker regression, runner ownership tests and native R5-002 follow-up", "result": "pass"}, + {"requirement": "full public pair/evaluate/qualify/reviewed promotion/new-run/held-out regression/rollback chain", "proof": "test_public_trials.PublicTrialTest.test_public_outcomes_evaluate_qualify_promote_new_run_and_roll_back", "result": "pass_offline_actual_workers"}, + {"requirement": "safe schema-changing upgrade and installed SDK transport", "proof": "full installer old schema13 active/recoverable deferral, cancellation/resume and migration-to16 tests; installed SDK nine tests", "result": "pass"} + ], + "installation": { + "release": "0.1.0-py31214-90f1e87fb9ac-mcp-a26bc88afbef", "source_digest": "90f1e87fb9acf7756dad46780672de9b84fe75e3631322857a310f089280345e", + "python": "3.12.14", "mcp": "2.2.0", "ledger_schema": 16, "nonterminal_runs_before": 0, "nonterminal_runs_after": 0, + "legacy_projection_jobs_after_migration": 0, "first_changed": true, "reinstall_changed": false, "payload_drift": false, "manifest_matches": true, "pip_check": "pass", "network_downloads": false, "previous_releases_retained": true, + "installed_sdk": {"tests": 9, "seconds": 2.674, "skips": 0, "failures": 0, "errors": 0, "origin": "resolved installed release, not source"}, + "normal_review_dry_run": {"model": "gpt-6.1-sol", "effort": "low", "selection_mode": "bounded_trial", "run_created": false, "task_sha256": "2f71ea5247b55aa4335e737f069b0d84d4ef13e7afc18d0dc9406e4f5185e1ab"}, + "matching_host_registrations": 4, + "doctor_note": "R5 doctor still describes installation/registration, not full authenticated or operation-verified readiness; R6 repairs this distinction" + }, + "other_gates": {"bash_assertions": 227, "generated_reference": "current", "git_diff_check": "pass"}, + "failure_history": "R5-public-integration-partial-2026-10-02.json preserves rejected native findings, red tests, interrupted and failed full invocations", + "limits": "Actual offline workers prove runtime/provenance, not model superiority or subscription savings. Automatic experimentation/promotion and Jev remain off. R6 usability, R7/C1 Council and R8 final desktop/live acceptance remain separate." +} From f343e25660a83b0d9499ef2c57328b327d39bee7 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 15:19:20 -0700 Subject: [PATCH 180/197] WIP checkpoint: R6 integration: terminal/readiness and shared probe cleanup; checkpoint independent finish-recovery finding (2026-10-02 15:19) --- docs/RUNTIME-GUIDE.md | 87 +++- docs/generated/core-reference.md | 5 +- docs/plans/engineering-team/RESUME.md | 61 ++- docs/plans/engineering-team/backlog.json | 6 +- ...terminal-readiness-partial-2026-10-02.json | 74 +++ plugin/core/src/devsquad/cli.py | 184 +++++++- plugin/core/src/devsquad/diagnostics.py | 254 +++++++++- plugin/core/src/devsquad/probe_process.py | 111 +++++ plugin/core/src/devsquad/service.py | 103 +++- plugin/core/src/devsquad/store.py | 13 + plugin/core/src/devsquad/task_entry.py | 30 +- test/core/fakes/claude_delivery_cli.py | 35 ++ test/core/fakes/codex_review_cli.py | 11 + test/core/test_cli.py | 35 +- test/core/test_diagnostics.py | 442 ++++++++++++++++++ test/core/test_installed_usability.py | 182 ++++++++ test/core/test_mcp.py | 6 +- test/core/test_task_entry.py | 124 ++++- test/core/test_terminal_ux.py | 282 +++++++++++ 19 files changed, 1963 insertions(+), 82 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json create mode 100644 plugin/core/src/devsquad/probe_process.py create mode 100755 test/core/fakes/claude_delivery_cli.py create mode 100644 test/core/test_diagnostics.py create mode 100644 test/core/test_installed_usability.py create mode 100644 test/core/test_terminal_ux.py diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index 477f302..7861695 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -82,9 +82,14 @@ profiles and embeds the validated routing snapshot. It does not require task, profile or policy JSON: ```bash -squad review --base main --wait --json +squad doctor +squad review --base main --dry-run +squad review --base main --wait squad fix "the bounded issue to resolve" \ - --write-path src --check "bash test/run.sh" --wait --json + --write-path src --wait +squad status +squad finish RUN_ID --accept --reason "Reviewed the saved candidate and evidence." +squad result RUN_ID ``` Use `--dry-run` first to inspect exact commit IDs, role/profile selections, @@ -92,7 +97,38 @@ scope and checks without creating a run or invoking a model. `squad fix` defaults to repository-wide write scope when `--write-path` is omitted; narrow it whenever the issue permits. `--wait` automatically advances the saved candidate from implementation into independent review, then returns at the -host-lead handoff. Omitting `--wait` returns the run ID immediately. +host-lead handoff with the review summary, check results, evidence report and +finish command. That pause exits 2; it is saved work awaiting your assessment. +Use `--reject` or `--revise` instead of `--accept` when appropriate, with a +reason. Finish binds every artifact from the exact current packet and applies +the same claim, independent-review and required-check gates as the low-level +API. It refuses terminal replays and existing host claims; use the saved claim +for a handoff already owned by an app. The normal lead is the terminal host; +an explicitly configured headless run follows its existing lead through a +temporary handoff when observed with `--wait`. Omitting `--wait` returns the run +ID immediately. + +Normal commands show readable output by default. Add `--json` for the versioned +automation envelope. Omitted IDs on status, result and finish resolve only when +the current canonical Git project has exactly one saved run. With no run, the +command explains how to start; with multiple runs it lists IDs and states and +requires a choice. It never selects another project's run or guesses the latest +run. Use `--project-dir PATH` when observing from outside the project. + +Checks come from regular tracked files at the selected target commit. For +DevSquad this includes the Bash suite, Python core runner and generated +reference check. Delivery requires all three to pass; review-only checks are +reported even when they fail. Explicit `--check` arguments add bounded approved +commands and exact duplicates run once. A directory called `tests` is not +enough to infer a Python suite without actual tracked Python test files. + +```mermaid +flowchart LR + A[review or fix] --> B[Exact candidate review and checks] + B --> C[Saved handoff and evidence] + C --> D[finish with accept, reject or revise] + D --> E[Saved receipt or bounded revision] +``` The lower-level automation path remains available. A hand-written task must name an existing Git repository and committed refs. It may use either committed @@ -135,9 +171,10 @@ returns the original run; reusing it with different content conflicts. Status is the authority for the current state and next action. Result artifacts and their SHA-256 values are authoritative; a chat summary is not. -When a host-lead workflow pauses, status returns `claim_handoff` and the -current run version. Claim and complete the saved packet without editing the -claim: +For terminal use, `squad finish` is the supported guided disposition command. +Automation and app hosts can still use the low-level claim/complete protocol. +When a host-lead workflow pauses, JSON status returns `claim_handoff` and the +current run version. Claim and complete the saved packet without editing the claim: ```bash squad handoff claim RUN_ID --expected-version VERSION --owner local-operator --json > claim-response.json @@ -160,7 +197,7 @@ Never delete the runtime database, an active release or a run-owned worktree to recover a job. Inspect first: ```bash -squad status RUN_ID --json +squad status RUN_ID squad events RUN_ID --after 0 --limit 100 --json ``` @@ -175,8 +212,8 @@ If status does not request recovery, do not invent a recovery decision. Cancellation is durable and idempotent: ```bash -squad cancel RUN_ID --json -squad status RUN_ID --json +squad cancel RUN_ID +squad status RUN_ID ``` For installation drift, run `scripts/install-core.sh --status --json` from the @@ -187,13 +224,14 @@ install a new source digest and retain the old directory for run evidence. ## Supported and deferred boundaries The current packaged contract is Python 3.11+, public JSON contract version 1, -SQLite schema 13 and optional MCP SDK 2.2.0 exactly. Native Codex fixtures and +SQLite schema 16 and optional MCP SDK 2.2.0 exactly. Native Codex fixtures and recorded live proofs cover bundled `codex-cli 0.153.4` and `codex-cli 0.155.0-alpha.9.2`; the latter passed a fresh native initialize and complete model-catalog probe. The resolver prefers that verified bundled binary over an older unverified PATH binary. The Claude worker adapter is version-scoped to CLI 2.1.220. Antigravity 1.2.13 has a live read-only MCP -receipt; Grok 0.2.111 has matching registration but expired authentication. +receipt; later installed CLI/MCP receipts cover Grok 1.0.46. Check the current +doctor report rather than treating an older receipt as proof for a new version. Version changes are capability drift and require a fresh conformance probe; brand names are not a compatibility promise. @@ -201,18 +239,25 @@ Implemented surfaces and evidence: | Surface | Current evidence | |---|---| -| Terminal | Standalone install plus a real saved-run start and terminal cancellation | +| Terminal | Standalone install and real saved-run cancellation; fresh installed normal review/fix/finish flow verified with offline provider binaries | | Codex App/CLI | Matching MCP registration and a fresh installed-runtime `squad_status` receipt on 0.155.0-alpha.9.2 | -| Claude Code local Code tab | Matching registration; live operation blocked on normal provider login | +| Claude Code local Code tab | Matching registration and real Claude MCP handoff; local Code-tab UI proof remains open | | Antigravity local IDE/CLI | Matching registration and a live Gemini `squad_status` receipt with one project-scoped grant | -| Grok Build | Matching registration; live operation blocked on expired authentication | +| Grok Build | Matching registration and real Grok 1.0.46 MCP status operation | -The current evidence source is +The historical installed surface evidence source is [`M7-installed-runtime-2026-09-29.json`](plans/engineering-team/evidence/M7-installed-runtime-2026-09-29.json). The installed normal-entry evidence is [`M7-normal-entry-2026-09-29.json`](plans/engineering-team/evidence/M7-normal-entry-2026-09-29.json). -Claude, Grok and the installed two-model delivery remain open, so universal -surface support is not yet claimed. +Later verified runtime proofs, including the accepted two-model delivery, +actual Claude handoff, Grok MCP and Gemini CLI/MCP recheck, are recorded in +[`R8-installed-workflows-2026-10-01.json`](plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json). +These are operation-scoped receipts; desktop UI proofs and the remaining R6/R8 +acceptance gates remain open. Doctor separates installed binaries, supported +adapter versions, non-generating authentication checks, registrations and +operation verification. Unknown verification stays unknown. A CLI that is +installed but unsupported or missing subscription authentication does not make +review/fix ready. Complete login through the named provider's normal flow. Portable task files, handoff packets, event ledgers and hashed artifacts are the cross-host interface. A reviewed upstream Codex integration demonstrates @@ -221,7 +266,7 @@ expose or test that import path. It is deferred and must never be substituted for the portable handoff contract or described as general chat-history transfer. -The optional Jev decision probe remains blocked until `TYPESAFE_API_KEY` is -provided; normal routing defaults to the deterministic zero-call path. Laya is -not installed unless the declared Jev fallback trigger fires. The optional C1 -Council extension is also deferred and does not block the core runtime. +The one authorized Jev pilot is recorded separately and runtime classification +remains off; normal routing uses the deterministic zero-call path. Laya is not +installed unless its declared trigger fires. C1 Council implementation and +acceptance remain pending in the full engineering-team delivery. diff --git a/docs/generated/core-reference.md b/docs/generated/core-reference.md index d0c8ee3..84e1483 100644 --- a/docs/generated/core-reference.md +++ b/docs/generated/core-reference.md @@ -12,6 +12,7 @@ Regenerate with `python3 scripts/generate-core-reference.py`; verify with - `squad classify [-h] [--cwd CWD] [--model MODEL] [--effort EFFORT] [--permission {read_only,workspace_write}] [--timeout TIMEOUT] [--transport {cli_exec,native_protocol}] [--catalog-file CATALOG_FILE] --returncode RETURNCODE --stdout-file STDOUT_FILE --stderr-file STDERR_FILE {codex,antigravity,grok}` - `squad doctor [-h] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` - `squad events [-h] [--after AFTER] [--limit LIMIT] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad finish [-h] (--accept | --reject | --revise) --reason REASON [--project-dir PROJECT_DIR] [--json] [--runtime-dir RUNTIME_DIR] [run]` - `squad fix [-h] [--base BASE] [--target TARGET] [--project-dir PROJECT_DIR] [--write-path WRITE_PATH] [--check CHECK] [--check-timeout CHECK_TIMEOUT] [--review-model REVIEW_MODEL] [--review-effort REVIEW_EFFORT] [--review-mode {standard,adversarial}] [--review-focus REVIEW_FOCUS] [--implementer-model IMPLEMENTER_MODEL] [--implementer-effort IMPLEMENTER_EFFORT] [--idempotency-key IDEMPOTENCY_KEY] [--dry-run] [--wait] [--json] [--runtime-dir RUNTIME_DIR] issue` - `squad handoff [-h] {claim,complete} ...` - `squad handoff claim [-h] --expected-version EXPECTED_VERSION --owner OWNER [--claim-file CLAIM_FILE] [--json] [--runtime-dir RUNTIME_DIR] run` @@ -33,12 +34,12 @@ Regenerate with `python3 scripts/generate-core-reference.py`; verify with - `squad profile qualification-add [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` - `squad profile template-add [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` - `squad report [-h] --project PROJECT [--json] [--runtime-dir RUNTIME_DIR]` -- `squad result [-h] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad result [-h] [--json] [--runtime-dir RUNTIME_DIR] [--project-dir PROJECT_DIR] [run]` - `squad resume [-h] [--json] [--runtime-dir RUNTIME_DIR] [--recovery-file RECOVERY_FILE] run` - `squad review [-h] [--base BASE] [--target TARGET] [--project-dir PROJECT_DIR] [--model MODEL] [--effort EFFORT] [--mode {standard,adversarial}] [--focus FOCUS] [--check CHECK] [--check-timeout CHECK_TIMEOUT] [--idempotency-key IDEMPOTENCY_KEY] [--dry-run] [--wait] [--json] [--runtime-dir RUNTIME_DIR]` - `squad setup [-h] [--host {codex,claude-code,antigravity,grok}] [--dry-run] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` - `squad start [-h] --task-file TASK_FILE --idempotency-key IDEMPOTENCY_KEY [--supersedes-run SUPERSEDES_RUN] [--wait] [--json] [--runtime-dir RUNTIME_DIR]` -- `squad status [-h] [--json] [--runtime-dir RUNTIME_DIR] run` +- `squad status [-h] [--json] [--runtime-dir RUNTIME_DIR] [--project-dir PROJECT_DIR] [run]` - `squad trial [-h] --experiment EXPERIMENT --case CASE --arm {control,candidate} --task-file TASK_FILE --idempotency-key IDEMPOTENCY_KEY [--wait] [--json] [--runtime-dir RUNTIME_DIR]` ## Common command examples diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 03db090..96d676e 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -72,15 +72,45 @@ do not repeat accepted proofs. No test/native process is active here. Next: integrate/test R6 terminal/readiness, R7/C1 and final R8 acceptance; whole-plan completion is not claimed. -R6 terminal and doctor slices are delegated in isolated attached worktrees -`r6-terminal` and `r6-readiness` at bf3de086. R7 Council is implementing in -isolated `r7-council` from the same baseline. They must not alter this frozen -candidate or production installation. Readiness checkpoint `dd709804` needs -a root-review follow-up for a SIGTERM-ignoring descendant after parent exit. -Integrate only clean checkpoint commits; preserve work if interrupted. macOS default- -deny sandbox probe establishes own-evidence access with peer/ledger denial, -not native Council compatibility or acceptance. No new native Council call -has been made. +R6 terminal/readiness checkpoints `dd709804`, `79f9b767`, `b785d040`, +`d0eec55c`, shared helper `4a6bc15d` and catalog reuse `03124c79` are now +integrated for the R6 source checkpoint in the main tree. +The fresh temporary install reaches review/fix receipts with only provider +binaries faked: actual normal discovery, distinct verified Claude/Codex +identity, required seeded Python check, guided finish, original checkout +unchanged, and project-scoped zero/one/multiple run choices. No task/decision +JSON or internal Service.start fixture is used. Agent gates and failures are +in `evidence/R6-terminal-readiness-partial-2026-10-02.json`; this is not yet +an accepted or installed R6 package. Root's combined focused invocation had +two nonexistent module names: the loader errors are not a passing gate or +product failures. Corrected affected/full gates remain required. + +Readiness `79f9b767` fixes the root-reviewed exited-parent/ignoring-descendant +cleanup defect. The same red reproduction in native catalog discovery is +repaired by the shared bounded helper; `03124c79` has 106 affected tests/ +95.408s with no skips/warnings, Bash and reference gates. Root's integrated +114-test gate (two optional SDK skips), 32 delivery/handoff tests and final +27 task-entry/fresh-install tests in 25.389s pass. Independent terminal review +found a real P1: interruption after guided finish acquires its claim but before +completion leaves no saved claim for retry; initial-only refuses it even after +expiry. A controlled offline reproduction confirms this, not a usage timeout. +The terminal agent is repairing durable exact intent/claim recovery without +app-claim takeover or changed disposition. Integrate the repair, then launch +one frozen full gate and exact R6 audit. No native/full gate or production +installation change has started for this candidate. +R7 Council remains in isolated `r7-council`; its controlled stage flow is +partial. Default-deny macOS own-evidence/peer-ledger denial and native binary +version probes are boundary mechanics, not a live Council receipt. No new +native Council generation has been made. Preserve all attached worktrees. + +Desktop control worked for scoped inspection. Claude's local Code tab selected only this +DevSquad project on `codex/engineering-team`, with an empty prompt; no proof +request was sent. The user clarified **Antigravity means CLI, not IDE**; +do not treat IDE trust/UI as a required Gemini acceptance gate. An IDE folder +chooser was inspected only: project trust, settings and MCP servers were not +changed. The cancellation attempt returned a new TCC capture denial, so UI +closure is unverified; do not bypass it. Recheck updated agy 1.2.14 CLI against +the accepted final release. Unrelated servers/settings remain untouched. Workspace: /Users/Dikshant/Desktop/Projects/devsquad. Branch: `codex/engineering-team`; never restart this build from main. @@ -117,13 +147,14 @@ integrity; 27 affected tests in 24.173s. Verified native Codex review is clean. 1. The user's three requested runtime actions are verified. Do not repeat these accepted proofs or the unchanged full suite. Both exact packets and artifact hashes are verified; never edit frozen evidence. -2. R3 and R4 are accepted by their October 2 closure matrices. Continue R5 - public terminal outcomes/trial controller, then remaining R6 UX/readiness, - R7/C1 and R8 whole-delivery acceptance. Preserve the existing reader and +2. R3/R4/R5 are accepted by their October 2 closure matrices. Finish integrated + R6 UX/readiness and its shared probe cleanup, then R7/C1 and R8 whole-delivery + acceptance. Preserve the existing reader and lifecycle eligibility gates; do not restart the architecture exercise. -3. Desktop UI proofs remain separate. Antigravity IDE control is permission- - denied; do not bypass it or substitute a CLI receipt. Grok/Gemini MCP status - calls do not prove automatic writer/reviewer adapters. +3. Claude Code-tab proof remains separate and unverified; current control + returned TCC denial, do not bypass it. Antigravity IDE is explicitly outside + the user's clarified request; Gemini acceptance uses the agy CLI. Grok/Gemini + MCP status calls do not prove automatic writer/reviewer adapters. Only one full core suite may run at a time; freeze source/tests while it runs. Use the tracked spawn-safe scripts/run-core-tests.py, never a stdin main. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index bd56022..b744ad2 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -22,7 +22,7 @@ "artifact": "evidence/R5-closure-2026-10-02.json", "scope": "R3/R4/R5 accepted; installed schema16 objective outcomes/public trials with 504-test full gate, exact native repair review, safe old-schema upgrade tests and nine installed SDK tests. Earlier accepted Claude delivery/handoff and Grok/Gemini receipts retained separately.", "next_action": "Integrate and verify delegated R6 terminal/readiness changes, then R7/C1 Council and final R8 acceptance. Do not repeat unchanged accepted proofs.", - "limitations": "Whole-plan closure is not claimed. Required desktop UI proofs remain unverified; Antigravity IDE permission is denied. Grok/Gemini MCP status does not prove automatic writer/reviewer roles. Jev remains off with its one-request allowance spent; no paid API fallback or reset use authorized." + "limitations": "Whole-plan closure is not claimed. Claude Code-tab proof remains unverified with latest TCC control denial. User clarified Antigravity CLI, not IDE; final updated agy CLI proof remains, not IDE trust. Grok/Gemini MCP status does not prove automatic writer/reviewer roles. Jev remains off with its one-request allowance spent; no paid API fallback or reset use authorized." }, "planning_checkpoint": { "recorded_on": "2026-10-01", @@ -47,8 +47,8 @@ {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "complete", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-closure-2026-10-02.json", "checkpoint": "Accepted dfe9976/source equivalent f87060b and installed schema16. Initial native findings repaired; exact follow-up e046a174 clean/accepted. Full 504 tests/511.241s, two optional SDK skips/no errors/failures/unraisable; nine actual installed SDK tests pass/no skips. Public complete trial/evaluate/qualify/promote/new-run/held-out regression/rollback, shared call/active deadline and all terminal projection origins pass. Safe old-schema upgrade and idempotence/drift gates pass. Failed history retained; automatic experimentation and Jev stay off."}, - {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"G4 discovery verified by accepted native 477-test delivery and combined review; integrated and installed at 6d2e0ba. Remaining UX and R4/R5 dependencies stay open."}, - {"id": "R7", "title": "Complete Council within existing runner", "status": "pending", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"]}, + {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Readiness, guided finish, human defaults, project-scoped run choices, committed-target checks and actual temporary installed normal review/fix fixtures integrated; shared catalog descendant cleanup repaired and focused root gates pass. Independent review found stranded guided-finish claim after interruption; repair required before full/native acceptance and installed gates. Current production remains accepted R5."}, + {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Implementation isolated in attached r7-council worktree. Controlled offline proposal/critic/lead flow and macOS boundary probes are partial; no accepted Council integration, comparison or native generation receipt yet."}, {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} ], "milestones": [ diff --git a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json new file mode 100644 index 0000000..9b7b73f --- /dev/null +++ b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json @@ -0,0 +1,74 @@ +{ + "schema_version": 1, + "work_package": "R6", + "recorded_on": "2026-10-02", + "status": "integrated_partial_repair_required", + "base_revision": "8f2c1c6a7d9edfe7cc3d78ecbc59ce105be636b2", + "integrated_checkpoints": [ + "dd709804e70e0791614319a9ae5c2597768d7318", + "79f9b767984f2eb64dca430941f405ef952ecf68", + "b785d04013042960b25b47292c3c3a5ead528c87", + "d0eec55c9dbede205e3c8831b7c224d3d77114d3", + "4a6bc15d81017aa996be9a21e6c81bc02ebd17a6", + "03124c79ba53cb1b12296bb554024605e4cadbc7" + ], + "contract": "SOL-REVIEW-FOLLOWUP R6/G3/G4; unchanged low-level MCP envelopes, fenced host disposition, Bash 3.2 and optional jq", + "implemented": [ + "Readable normal terminal commands; --json retains the versioned automation envelope", + "Guided finish binds the exact current packet/artifact hashes and existing independent-review/check/claim fences; rejects replay and prior host claims", + "Omitted IDs require exactly one canonical current-project run; zero and multiple choices are explicit; unrelated runs never selected", + "Doctor distinguishes installed, registered, exact supported version, subscription authentication, workflow readiness and unknown operation proof", + "Bounded non-generating Claude auth and Codex account/read with no API keys or alternate credential-home overrides", + "Committed target check discovery includes Bash, actual Python core tests and generated reference; exact argv dedup and byte/NUL/regular-blob safety retained", + "Fresh source install, actual installed normal review/fix discovery and guided receipt path with only native provider binaries replaced by offline fixtures" + ], + "agent_gates": [ + {"revision": "dd709804", "tests": 37, "seconds": 7.852, "optional_sdk_skips": 2, "bash_assertions": 227}, + {"revision": "79f9b767", "tests": 41, "seconds": 10.391, "optional_sdk_skips": 2, "bash_assertions": 227}, + {"revision": "b785d040", "tests": 103, "seconds": 89.115, "skips": 0, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"}, + {"revision": "d0eec55c", "tests": 1, "seconds": 11.955, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"}, + {"revision": "4a6bc15d", "tests": 41, "seconds": 11.356, "optional_sdk_skips": 2, "bash_assertions": 227}, + {"revision": "03124c79", "tests": 106, "seconds": 95.408, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"} + ], + "root_affected_gates": [ + {"command": "python3 -m unittest test_diagnostics test_terminal_ux test_installed_usability test_cli test_task_entry test_mcp test_handoff_store", "tests": 114, "seconds": 65.066, "optional_sdk_skips": 2, "failures": 0, "errors": 0, "scope": "Integrated first five checkpoints, before final catalog cleanup reuse"}, + {"command": "python3 -m unittest test_delivery_workflow test_delivery_identity_runtime test_handoff_service", "tests": 32, "seconds": 66.877, "failures": 0, "errors": 0}, + {"command": "python3 -m unittest test_task_entry test_installed_usability", "tests": 27, "seconds": 25.389, "skips": 0, "failures": 0, "errors": 0, "scope": "All six integrated checkpoints including final catalog cleanup reuse"} + ], + "failure_history": [ + "Readiness red baseline: nine tests, three failures and 17 subtest errors before implementation", + "Terminal red baseline: six tests, four failures and two errors before implementation", + "Root review reproduced a surviving SIGTERM-ignoring descendant after the diagnostic parent exited; repaired by 79f9b767 with process-group absence/ownership checks and bounded escalation", + "Catalog discovery independently reproduced the same surviving-descendant defect (one failed test/0.133s); shared helper reuse in 03124c79 passes three focused tests/1.004s plus 106 affected tests. Root final gate still required", + "Fresh installed fixture seeded check wrote original-checkout bytecode; PYTHONDONTWRITEBYTECODE=1 restored the declared pristine fixture contract without weakening source gates", + "Root integrated affected command ran 116 tests in 69.606s, two optional SDK skips and two loader errors from nonexistent test_review_service/test_delivery_service module names; not a passing command gate and not a runtime defect", + "A separate follow-up invocation mistakenly included nonexistent test_delivery_runtime; corrected existing-module/full gates are required", + "Independent bounded terminal review reproduced a P1: interruption after guided finish claim commit before completion strands the run without saved claim; same finish retry rejects both live and expired own claim. Durable exact-intent recovery is being repaired before full/native acceptance" + ], + "fresh_install_scope": { + "providers": "offline native CLI protocol fixtures only", + "installer": "actual source installer with no-index and temporary HOME/install/runtime/bin", + "internal_fixture_injection": false, + "manual_task_json": false, + "manual_decision_json": false, + "discovery_stub": false, + "verified_distinct_writer_reviewer": true, + "seeded_python_check": "fails original and passes candidate", + "original_head_source_porcelain": "unchanged", + "receipt_artifact_hashes": "verified", + "provider_generation_or_production_update": false + }, + "desktop_scope": { + "control": "available for initial scoped inspection; last IDE cancel attempt returned TCC denial, not bypassed", + "claude": "Local Code tab selected DevSquad on codex/engineering-team; empty prompt; no proof call sent", + "antigravity": "User clarified Antigravity CLI, not IDE. No IDE trust/settings/MCP edit was performed; folder chooser only inspected, cancel closure unverified after TCC denial. Updated agy CLI requires final-release status proof", + "unrelated_servers_or_settings": "not changed" + }, + "remaining_gates": [ + "Guided finish interrupted claim/intent recovery repair and adversarial regressions", + "Frozen full gate", + "Independent exact R6 review and accepted immutable checkpoint", + "Safe installed update, non-generating live readiness, idempotence/drift and actual installed SDK tests", + "Separate Claude Code-tab if accessible, updated Antigravity CLI and R7/R8 gates; Antigravity IDE not required by user clarification" + ] +} diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index bfee1e4..daefbfb 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -5,6 +5,7 @@ import argparse import json import os +import shlex import sys import time from pathlib import Path @@ -35,6 +36,7 @@ "cancelled": 4, } WAIT_ACTIVE_STATES = {"queued", "running", "cancelling"} +HUMAN_COMMANDS = {"review", "fix", "status", "result", "doctor", "setup", "finish", "resume", "cancel"} def manifests() -> list[tuple[Path, AdapterManifest]]: @@ -145,11 +147,24 @@ def _wait_for_run( while True: status = service.status(run_id) state = status.get("state") + version = status.get("version") + if (state == "awaiting_host" + and status.get("next_action") == "continue_headless_lead"): + if type(version) is int and version not in resumed_versions: + resumed_versions.add(version) + try: + service.resume(run_id) + except ConflictError: + # The detached owner may have continued the same + # headless handoff between observation and resume. + if service.status(run_id).get("version") == version: + raise + time.sleep(WAIT_POLL_SECONDS) + continue if state in WAIT_EXIT_CODES: return envelope(data=status), WAIT_EXIT_CODES[state] if state not in WAIT_ACTIVE_STATES: raise RuntimeError(f"service returned unsupported run state: {state!r}") - version = status.get("version") if (resume_candidate_review and state == "queued" and status.get("next_action") == "resume_candidate_review" @@ -297,6 +312,8 @@ def _command_normal_entry( else (envelope(data=started), 0) ) service_data = run["data"] + if not args.json: + service_data = _display_status(service, service_data) return envelope(data=_normal_entry_result( summary, idempotency_key, service_data, )), code @@ -372,15 +389,43 @@ def command_profile_binding_show(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).profile_binding_status(args.alias)), 0 -def command_status(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).status(args.run)), 0 +def _selected_run(args: argparse.Namespace, service: Service) -> str: + return args.run if args.run is not None else service.resolve_run_id(None, Path(args.project_dir)) + + +def _display_status(service: Service, data: dict[str, Any]) -> dict[str, Any]: + if data.get("state") == "awaiting_host" and (data.get("handoff") or {}).get("status") == "open": + try: + view = service.handoff_view(data["run_id"]) + except ConflictError as exc: + # A live headless lead can advance while its status is rendered. + return {**data, "handoff_view_error": str(exc)} + if view["version"] == data.get("version"): + return {**data, "handoff_view": view} + return data + + +def command_status(args: argparse.Namespace) -> tuple[dict, int]: + service = _service(args) + data = service.status(_selected_run(args, service)) + return envelope(data=data if args.json else _display_status(service, data)), 0 def command_events(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).events(args.run, args.after, args.limit)), 0 -def command_result(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).result(args.run)), 0 +def command_result(args: argparse.Namespace) -> tuple[dict, int]: + service = _service(args) + return envelope(data=service.result(_selected_run(args, service))), 0 def command_cancel(args: argparse.Namespace) -> tuple[dict, int]: return envelope(data=_service(args).cancel(args.run)), 0 def command_resume(args: argparse.Namespace) -> tuple[dict, int]: recovery = _read_json(args.recovery_file, "recovery file") if args.recovery_file else None return envelope(data=_service(args).resume(args.run, recovery)), 0 +def command_finish(args: argparse.Namespace) -> tuple[dict, int]: + service = _service(args) + return envelope(data=service.finish( + _selected_run(args, service), args.disposition, args.reason, + )), 0 + + def command_handoff_claim(args: argparse.Namespace) -> tuple[dict, int]: prior_claim = _read_json(args.claim_file, "claim file") if args.claim_file else None return envelope(data=_service(args).handoff_claim( @@ -512,9 +557,20 @@ def parser() -> argparse.ArgumentParser: trial.add_argument("--runtime-dir", default=runtime_default) trial.set_defaults(func=command_trial) for name, fn in (("status",command_status),("result",command_result),("cancel",command_cancel),("resume",command_resume)): - cmd=sub.add_parser(name); cmd.add_argument("run"); cmd.add_argument("--json",action="store_true"); cmd.add_argument("--runtime-dir",default=runtime_default) + cmd=sub.add_parser(name); cmd.add_argument("run", nargs="?" if name in {"status", "result"} else None); cmd.add_argument("--json",action="store_true"); cmd.add_argument("--runtime-dir",default=runtime_default) + if name in {"status", "result"}: cmd.add_argument("--project-dir", default=str(Path.cwd())) if name == "resume": cmd.add_argument("--recovery-file") cmd.set_defaults(func=fn) + finish = sub.add_parser("finish", help="decide the current host handoff without decision JSON") + finish.add_argument("run", nargs="?") + disposition = finish.add_mutually_exclusive_group(required=True) + for value in ("accept", "reject", "revise"): + disposition.add_argument(f"--{value}", dest="disposition", action="store_const", const=value) + finish.add_argument("--reason", required=True) + finish.add_argument("--project-dir", default=str(Path.cwd())) + finish.add_argument("--json", action="store_true") + finish.add_argument("--runtime-dir", default=runtime_default) + finish.set_defaults(func=command_finish) events=sub.add_parser("events"); events.add_argument("run"); events.add_argument("--after",type=int,default=0); events.add_argument("--limit",type=int,default=100); events.add_argument("--json",action="store_true"); events.add_argument("--runtime-dir",default=runtime_default); events.set_defaults(func=command_events) capacity = sub.add_parser("capacity") capacity_sub = capacity.add_subparsers(dest="capacity_command", required=True) @@ -598,22 +654,132 @@ def parser() -> argparse.ArgumentParser: return p +def _next_command(data: dict[str, Any]) -> str: + run_id = data.get("run_id", "RUN") + action = data.get("next_action") + if action == "claim_handoff": + owner = (data.get("handoff") or {}).get("claimed_by") + if owner: + return f"complete or renew the saved claim in {owner}; this handoff already has an owner" + return f'squad finish {run_id} --accept --reason "your assessment of the saved evidence" (or --reject / --revise)' + if action in {"continue_headless_lead", "resume_candidate_review", "handoff_submission_saved"}: + return f"squad resume {run_id}" + if action == "recovery_file_required": + return f"squad status {run_id} --json; inspect the recovery evidence before squad resume {run_id} --recovery-file FILE" + if data.get("state") in {"succeeded", "failed", "cancelled"}: + return f"squad result {run_id}" + if isinstance(action, str) and action: + return action.removesuffix(" --json") + return f"squad status {run_id}" + + +def _handoff_lines(data: dict[str, Any]) -> list[str]: + view = data.get("handoff_view") + if not view: + return [f"Evidence unavailable: {data['handoff_view_error']}"] if data.get("handoff_view_error") else [] + review = view["review"] + lines = [f"Review: {review['verdict']} — {review['summary']}"] + for finding in review.get("findings", []): + lines.append(f" {finding['severity']}: {finding['title']} ({finding['path']}:{finding['start_line']})") + for check in view["checks"]: + lines.append(f"Check {check['id']}: {check['status']}") + lines.append(f"Evidence {view['report']['name']}: {view['report']['path']}") + return lines + + +def _human_response(command: str, response: dict[str, Any]) -> str: + if not response["ok"]: + error = response["error"] + return f"{error['code']}: {error['message']}\nThe command did not complete. Existing runs remain saved; inspect squad status RUN." + data = response["data"] + lines = [] + if command == "doctor": + lines.append(f"DevSquad {data.get('core_version', __version__)}: {'ready' if data.get('ready') else 'needs attention'}") + for row in data.get("adapters", []): + auth = row.get("authentication", {}) + verified = row.get("operation_verified") + lines.append( + f"{row['adapter']}: {row.get('version') or 'not installed'}; " + f"adapter {row.get('status', 'unknown')}; " + f"authentication {auth.get('status', 'unknown')}; " + f"operation {'verified' if verified is True else 'unverified' if verified is False else 'unknown'}" + ) + if auth.get("next_action"): + lines.append(f" Next: {auth['next_action']}") + for workflow, row in data.get("supported_workflows", {}).items(): + lines.append(f"{workflow}: {'ready' if row.get('ready') else 'unavailable' if not row.get('supported') else 'needs attention'}") + for row in data.get("local_apps", []): + lines.append(f"{row.get('id', 'app')} registration: {row.get('status', 'unknown')}") + lines.append("Next: squad setup --dry-run" if not data.get("ready") else "Next: squad review --base main --dry-run") + elif command == "setup": + lines.append(f"Setup {'preview' if data.get('dry_run') else 'completed' if data.get('completed') else 'needs attention'}") + for row in data.get("hosts", []): + lines.append(f"{row.get('id', 'app')}: {row.get('action', 'unknown')}; registration {'ready' if row.get('ready') else 'needs attention'}") + lines.append("Next: squad setup" if data.get("dry_run") and data.get("completed") else "Next: squad doctor") + elif command in {"review", "fix"}: + lines.append(f"{data.get('workflow', command)}: {data.get('state', 'unknown')}") + if data.get("run_id"): + lines.append(f"Run: {data['run_id']}") + lines.append(f"Project: {data.get('project', 'unknown')}") + lines.append(f"Commits: {data.get('base_oid', '')} → {data.get('target_oid', '')}") + for role, row in data.get("planned_roles", {}).items(): + lines.append(f"{role}: {row.get('harness', 'unknown')} {row.get('model_id', '')} / {row.get('effort', 'unknown')} ({row.get('selection_mode', 'unknown')})") + if data.get("selection_reason"): + lines.append(f"Selection: {data['selection_reason']}") + scope = data.get("scope", {}) + lines.append(f"Read scope: {', '.join(scope.get('read_paths', []))}; write scope: {', '.join(scope.get('write_paths', [])) or 'none'}") + for check in data.get("check_plan", []): + lines.append(f"Check {check['id']}: {shlex.join(check['argv'])} ({'required' if check['required_to_pass'] else 'report only'})") + if not data.get("check_plan"): + lines.append(f"Checks: {', '.join(data.get('checks', []))}") + next_data = data.get("service") or data + lines.extend(_handoff_lines(next_data)) + lines.append(f"Next: {_next_command(next_data)}") + else: + run_id = data.get("run_id", "unknown") + lines.append(f"Run {run_id}: {data.get('state', 'unknown')}") + if data.get("phase"): + lines.append(f"Progress: {data['phase']}") + if command == "result": + if not data.get("ready"): + lines.append("Result is not ready; the run is saved.") + for artifact in data.get("artifacts", []): + lines.append(f"{artifact['name']}: {artifact['path']}") + if data.get("active_attempt"): + lines.append(f"Worker: {data['active_attempt'].get('status', 'unknown')}") + lines.extend(_handoff_lines(data)) + if data.get("disposition"): + lines.append(f"Disposition: {data['disposition']}") + if command == "cancel" and data.get("state") == "cancelling": + lines.append("Cancellation is saved; worker cleanup is still running.") + if command != "result" or not data.get("ready"): + lines.append(f"Next: {_next_command(data)}") + return "\n".join(lines) + + def main(argv: list[str] | None = None) -> int: + arguments = list(sys.argv[1:] if argv is None else argv) + command = arguments[0] if arguments else "" + def emit(response): + if command in HUMAN_COMMANDS and "--json" not in arguments: + print(_human_response(command, response)) + else: + print(json.dumps(response, sort_keys=True)) try: - args = parser().parse_args(argv) + args = parser().parse_args(arguments) if hasattr(args, "stream_func"): return args.stream_func(args) response, code = args.func(args) - print(json.dumps(response, sort_keys=True)) + emit(response) return code except (ConflictError, SchemaVersionError) as exc: - print(json.dumps(envelope(error=error_payload(exc.code, str(exc))), sort_keys=True)) + emit(envelope(error=error_payload(exc.code, str(exc)))) return 75 except ContractError as exc: - print(json.dumps(envelope(error=error_payload(exc.code, str(exc))), sort_keys=True)) + emit(envelope(error=error_payload(exc.code, str(exc)))) return 64 except Exception as exc: - print(json.dumps(envelope(error=error_payload("INTERNAL_ERROR", str(exc))), sort_keys=True)) + emit(envelope(error=error_payload("INTERNAL_ERROR", str(exc)))) return 1 diff --git a/plugin/core/src/devsquad/diagnostics.py b/plugin/core/src/devsquad/diagnostics.py index a58edf0..128a7c4 100644 --- a/plugin/core/src/devsquad/diagnostics.py +++ b/plugin/core/src/devsquad/diagnostics.py @@ -2,16 +2,33 @@ from __future__ import annotations +import json +import os from pathlib import Path +import re +import selectors +import shutil +import subprocess import sys +import time from typing import Any from . import __version__ -from .adapters import AdapterManifest, harness_version +from .adapters import AdapterManifest +from .codex_protocol import ( + JsonLinePeer, initialize_request, initialized_notification, + receive_response, request, +) +from .contracts import ContractError from .integrations import ( LocalIntegrationManager, load_integrations, ) +from .probe_process import ( + capture_probe_identity, + close_probe as _close_probe, + subscription_environment as _environment, +) SOURCE_ROOT = Path(__file__).resolve().parents[2] CORE_ROOT = ( @@ -19,19 +36,203 @@ if (SOURCE_ROOT / "adapters").is_dir() else Path(sys.prefix) / "share" / "devsquad" ) +PROBE_TIMEOUT_SECONDS = 3 +AUTH_TIMEOUT_SECONDS = 5 +MAX_PROBE_BYTES = 16 * 1024 +VERSION_PATTERN = re.compile( + r"(?:codex-cli )?\d+\.\d+\.\d+(?:-[A-Za-z0-9.]+)?(?: \(Claude Code\))?" +) + + +def _probe_output( + argv: list[str], *, project: Path, environment: dict[str, str], +) -> tuple[int, str]: + """Bound time and bytes before decoding any provider-controlled output.""" + deadline = time.monotonic() + PROBE_TIMEOUT_SECONDS + process = subprocess.Popen( + argv, cwd=project, env=environment, stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, start_new_session=True, + ) + start_identity = capture_probe_identity(process) + try: + assert process.stdout is not None + chunks = bytearray() + with selectors.DefaultSelector() as selector: + selector.register(process.stdout, selectors.EVENT_READ) + while True: + remaining = deadline - time.monotonic() + if remaining <= 0 or not selector.select(remaining): + raise TimeoutError("diagnostic probe timed out") + chunk = os.read(process.stdout.fileno(), min(4096, MAX_PROBE_BYTES + 1 - len(chunks))) + if not chunk: + break + chunks.extend(chunk) + if len(chunks) > MAX_PROBE_BYTES: + raise ContractError("diagnostic probe exceeds byte bound") + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("diagnostic probe timed out") + return process.wait(timeout=remaining), chunks.decode("utf-8") + finally: + _close_probe(process, start_identity=start_identity) + + +def _resolve_adapter( + manifest: AdapterManifest, *, project: Path, environment: dict[str, str], +) -> tuple[str | None, str | None]: + candidates = tuple(dict.fromkeys( + binary for name in manifest.binary_candidates + if (binary := shutil.which(name, path=environment["PATH"])) is not None + )) + first = (None, None) + for binary in candidates: + try: + code, output = _probe_output([binary, "--version"], project=project, environment=environment) + version = output.strip() if code == 0 else None + # A version banner can contain a login error or secrets. Only a + # bounded canonical version is safe to retain in the public report. + if version is not None and VERSION_PATTERN.fullmatch(version) is None: + version = None + except (ContractError, OSError, TimeoutError, UnicodeError, subprocess.TimeoutExpired): + version = None + if first[0] is None: + first = (binary, version) + if version in manifest.verified_versions: + return binary, version + return first + + +def _authentication( + *, authenticated: bool | None = None, method: str | None = None, + subscription_supported: bool = False, check: str | None = None, + reason: str | None = None, next_action: str | None = None, +) -> dict[str, Any]: + return { + "status": ( + "authenticated" if authenticated is True + else "unauthenticated" if authenticated is False else "unknown" + ), + "authenticated": authenticated, + "method": method, + "subscription_supported": subscription_supported, + "check": check, + "reason": reason, + "next_action": next_action, + } -def _adapter_rows() -> list[dict[str, Any]]: +def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + value: dict[str, Any] = {} + for key, item in pairs: + if key in value: + raise ValueError("duplicate diagnostic field") + value[key] = item + return value + + +def _claude_auth(binary: str, *, project: Path, environment: dict[str, str]) -> dict[str, Any]: + check = "claude auth status --json" + try: + code, output = _probe_output( + [binary, "auth", "status", "--json"], project=project, environment=environment, + ) + value = json.loads(output, object_pairs_hook=_unique_object) + if not isinstance(value, dict) or type(value.get("loggedIn")) is not bool: + raise ValueError("invalid authentication state") + logged_in, method = value["loggedIn"], value.get("authMethod") + if not isinstance(method, str) or method not in {"none", "claude.ai", "oauth_token", "api_key", "api_key_helper", "third_party"}: + raise ValueError("unknown authentication method") + if (code != (0 if logged_in else 1) + or (logged_in and method == "none") + or (not logged_in and method != "none")): + raise ValueError("inconsistent authentication state") + except (ContractError, OSError, TimeoutError, UnicodeError, ValueError, RecursionError, subprocess.TimeoutExpired): + return _authentication(check=check, reason="auth_check_failed") + subscription_supported = logged_in and method == "claude.ai" + return _authentication( + authenticated=logged_in, method=method, check=check, + subscription_supported=subscription_supported, + reason=(None if subscription_supported else "subscription_login_required"), + next_action=None if subscription_supported else "claude auth login", + ) + + +def _codex_auth(binary: str, *, project: Path, environment: dict[str, str]) -> dict[str, Any]: + check = "codex account/read" + deadline = time.monotonic() + AUTH_TIMEOUT_SECONDS + process = None + start_identity = None + try: + process = subprocess.Popen( + [binary, "app-server", "--listen", "stdio://"], + cwd=project, env=environment, stdin=subprocess.PIPE, + stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, + bufsize=1, start_new_session=True, + ) + start_identity = capture_probe_identity(process) + assert process.stdin is not None and process.stdout is not None + peer = JsonLinePeer(process.stdout, process.stdin, max_frame_bytes=MAX_PROBE_BYTES) + peer.send(initialize_request(1)) + initialized = receive_response(peer, 1, timeout_seconds=max(0.001, deadline - time.monotonic())) + if "error" in initialized or not isinstance(initialized.get("result"), dict): + raise ContractError("native initialization unavailable") + peer.send(initialized_notification()) + peer.send(request(2, "account/read", {"refreshToken": False})) + reply = receive_response(peer, 2, timeout_seconds=max(0.001, deadline - time.monotonic())) + value = reply.get("result") + if ("error" in reply or not isinstance(value, dict) or "account" not in value + or type(value.get("requiresOpenaiAuth")) is not bool): + raise ContractError("native authentication unavailable") + account = value["account"] + if account is None: + return _authentication( + authenticated=False if value["requiresOpenaiAuth"] else None, + check=check, reason="subscription_login_required", + next_action="codex login", + ) + if (not isinstance(account, dict) or not isinstance(account.get("type"), str) + or account["type"] not in {"chatgpt", "apiKey"}): + raise ContractError("native authentication method unknown") + method = account["type"] + subscription_supported = method == "chatgpt" and value["requiresOpenaiAuth"] + return _authentication( + authenticated=True, method=method, check=check, + subscription_supported=subscription_supported, + reason=None if subscription_supported else "subscription_login_required", + next_action=None if subscription_supported else "codex login", + ) + except (ContractError, EOFError, OSError, TimeoutError, ValueError, RecursionError, subprocess.TimeoutExpired): + return _authentication(check=check, reason="auth_check_failed") + finally: + if process is not None: + _close_probe(process, start_identity=start_identity) + + +def _adapter_rows(*, project: Path, home: Path | None = None) -> list[dict[str, Any]]: + environment = _environment(home) rows = [] for path in sorted((CORE_ROOT / "adapters").glob("*/adapter.json")): manifest = AdapterManifest.load(path) - binary = manifest.resolve_binary() - version = harness_version(binary) if binary else None + binary, version = _resolve_adapter(manifest, project=project, environment=environment) + supported = binary is not None and version in manifest.verified_versions status = ( "unavailable" if not binary - else "supported" if version in manifest.verified_versions + else "supported" if supported else "unverified" ) + if not binary: + authentication = _authentication(reason="not_installed") + elif manifest.name not in {"codex", "claude"}: + authentication = _authentication(reason="auth_check_unsupported") + elif not supported: + authentication = _authentication(reason="unverified_version") + else: + probe = _codex_auth if manifest.name == "codex" else _claude_auth + try: + authentication = probe(binary, project=project, environment=environment) + except (ContractError, OSError, subprocess.TimeoutExpired): + authentication = _authentication(reason="auth_cleanup_unconfirmed") + ready = supported and authentication["subscription_supported"] rows.append({ "adapter": manifest.name, "transport": manifest.transport, @@ -39,6 +240,17 @@ def _adapter_rows() -> list[dict[str, Any]]: "binary": binary, "version": version, "manifest": str(path), + "installed": binary is not None, + "supported": supported, + "authenticated": authentication["authenticated"], + "authentication": authentication, + "ready": ready, + # Authentication and MCP registration prove neither worker + # execution nor a compatible model/permission operation receipt. + "operation_verified": None, + "operation_verification": { + "status": "unknown", "reason": "no_compatible_operation_proof", + }, }) return rows @@ -52,17 +264,36 @@ def build_doctor_report( ) -> dict[str, Any]: """Report provider and installed-app readiness without modifying config.""" - adapters = _adapter_rows() + adapters = _adapter_rows(project=project, home=home) integration_manager = manager or LocalIntegrationManager( project=project, home=home, squad_executable=squad_executable, ) - local_apps = [ - integration_manager.inspect(template) for template in load_integrations() - ] + local_apps = [] + for template in load_integrations(): + row = integration_manager.inspect(template) + local_apps.append({ + **row, + "registered": row.get("status") == "matching", + "operation_verified": None, + }) installed_apps = [row for row in local_apps if row["installed"]] - adapter_ready = any(row["status"] != "unavailable" for row in adapters) + adapter_ready = any(row["ready"] for row in adapters) + by_adapter = {row["adapter"]: row for row in adapters} + supported_workflows = {} + for workflow, required in ( + ("branch-review", ("codex",)), ("issue-delivery", ("claude", "codex")), + ): + blocked = [name for name in required if not by_adapter.get(name, {}).get("ready")] + supported_workflows[workflow] = { + "supported": True, "ready": not blocked, + "required_adapters": list(required), "blocked_adapters": blocked, + } + supported_workflows["council"] = { + "supported": False, "ready": False, "reason": "not_implemented", + } + workflow_ready = any(row["ready"] for row in supported_workflows.values()) local_apps_required = bool(installed_apps) local_apps_ready = ( all(row["ready"] for row in installed_apps) @@ -70,8 +301,9 @@ def build_doctor_report( ) return { "core_version": __version__, - "ready": adapter_ready and local_apps_ready, + "ready": workflow_ready and local_apps_ready, "adapter_ready": adapter_ready, + "supported_workflows": supported_workflows, "local_app_access": { "required": local_apps_required, "ready": local_apps_ready, diff --git a/plugin/core/src/devsquad/probe_process.py b/plugin/core/src/devsquad/probe_process.py new file mode 100644 index 0000000..72f81dc --- /dev/null +++ b/plugin/core/src/devsquad/probe_process.py @@ -0,0 +1,111 @@ +"""Bounded ownership and cleanup for nongenerating native provider probes.""" + +from __future__ import annotations + +import getpass +import os +from pathlib import Path +import signal +import subprocess +import time +from typing import Any + +from .contracts import ContractError +from .supervisor import process_start_identity + +PROBE_CLEANUP_SECONDS = 2 +PROBE_TERM_GRACE_SECONDS = 0.25 + + +def subscription_environment(home: Path | None = None) -> dict[str, str]: + """Use saved subscription login without ambient keys/provider overrides.""" + return { + "HOME": str(home if home is not None else Path.home()), + "USER": os.environ.get("USER") or getpass.getuser(), + "PATH": os.environ.get("PATH", ""), + } + + +def capture_probe_identity(process: subprocess.Popen[Any]) -> str | None: + """Capture immediately after spawning with start_new_session=True.""" + return process_start_identity(process.pid) + + +def _probe_group_exists(pgid: int, *, deadline: float) -> bool: + """The supervisor's live-member rule with a diagnostic time bound.""" + remaining = deadline - time.monotonic() + if remaining <= 0: + raise ContractError("diagnostic cleanup deadline expired") + try: + inventory = subprocess.run( + ["/bin/ps", "-axo", "pgid=,stat="], text=True, + stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, + env=subscription_environment(), timeout=min(0.5, remaining), check=False, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise ContractError("diagnostic process inventory unavailable") from exc + if inventory.returncode != 0 or len(inventory.stdout) > 1024 * 1024: + raise ContractError("diagnostic process inventory unavailable") + parsed = 0 + for line in inventory.stdout.splitlines(): + fields = line.split() + if len(fields) != 2 or not fields[0].isdigit(): + raise ContractError("diagnostic process inventory malformed") + parsed += 1 + if int(fields[0]) == pgid and not fields[1].startswith("Z"): + return True + if parsed == 0: + raise ContractError("diagnostic process inventory empty") + return False + + +def close_probe(process: subprocess.Popen[Any], *, start_identity: str | None) -> None: + """Stop the owned group, confirm no live members, reap and close streams. + + The process must have been spawned by the caller with a new session and + its identity captured before protocol use. Inspection/ownership failures + raise rather than pretending cleanup or readiness was confirmed. + """ + deadline = time.monotonic() + PROBE_CLEANUP_SECONDS + def owned_group_exists() -> bool: + if not _probe_group_exists(process.pid, deadline=deadline): + return False + observed = process_start_identity(process.pid) + if observed is not None and observed != start_identity: + raise ContractError("diagnostic process identity changed before cleanup") + if observed is None and start_identity is None and process.returncode is not None: + raise ContractError("diagnostic process ownership unavailable") + return True + + def signal_group(value: int) -> None: + # Recheck immediately before each signal. A recycled leader PID is + # never authority to signal a new group. + if owned_group_exists(): + try: + os.killpg(process.pid, value) + except ProcessLookupError: + pass + + try: + if owned_group_exists(): + signal_group(signal.SIGTERM) + grace = min(deadline, time.monotonic() + PROBE_TERM_GRACE_SECONDS) + # Keep an exited direct child unreaped while descendants survive; + # its PID remains an ownership anchor through escalation. + while owned_group_exists() and time.monotonic() < grace: + time.sleep(min(0.02, max(0, grace - time.monotonic()))) + if owned_group_exists(): + signal_group(signal.SIGKILL) + while owned_group_exists(): + remaining = deadline - time.monotonic() + if remaining <= 0: + raise ContractError("diagnostic process group survived cleanup") + time.sleep(min(0.02, remaining)) + remaining = deadline - time.monotonic() + if remaining <= 0: + raise ContractError("diagnostic cleanup deadline expired") + process.wait(timeout=remaining) + finally: + for stream in (process.stdin, process.stdout, process.stderr): + if stream is not None: + stream.close() diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 792d55b..287f7a7 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -1292,6 +1292,7 @@ def status(self, run_id: str) -> dict[str, Any]: "resume_candidate_review" if snapshot.get("task", {}).get("workflow") == "issue-delivery" and isinstance(snapshot.get("candidate"), dict) + and (handoff is None or handoff.status != "open") else None ) except ConflictError: @@ -1317,6 +1318,102 @@ def status(self, run_id: str) -> dict[str, Any]: return result finally: store.close() + def resolve_run_id(self, run_id: str | None, project: Path) -> str: + """Only infer a run when this canonical Git project has one saved choice.""" + if run_id is not None: + return run_id + store = self._store() + try: + choices = store.runs_for_project(project) + finally: + store.close() + if not choices: + raise ContractError("No saved runs for this Git project. Start with squad review --base main or squad fix \"the bounded issue\".") + if len(choices) == 1: + return choices[0]["run_id"] + listed = "; ".join(f"{item['run_id']} ({item['state']})" for item in choices[:20]) + more = "; additional runs omitted" if len(choices) > 20 else "" + raise ConflictError(f"Multiple saved runs for this Git project; specify RUN: {listed}{more}") + + def handoff_view(self, run_id: str) -> dict[str, Any]: + """Read the current validated review and its portable human report.""" + from .reports import handoff_report_names + + store = self._store() + try: + run, _, handoff = store.status_snapshot(run_id) + if handoff is None or handoff.status != "open": + raise ConflictError("run has no current open handoff to inspect") + packet = validate_saved_review_handoff(handoff.packet, self._review_snapshot(run)) + report = store.artifact_named(run_id, handoff_report_names(handoff.sequence)[1]) + if report is None: + raise ConflictError("saved handoff report is missing") + try: + content = Path(report["path"]).read_bytes() + except OSError as exc: + raise ConflictError("saved handoff report is missing") from exc + if (len(content) != report["byte_size"] + or hashlib.sha256(content).hexdigest() != report["sha256"]): + raise ConflictError("saved handoff report is corrupt") + return { + "version": run["version"], "packet_sha256": handoff.packet_sha256, + "candidate_sha256": packet["candidate_sha256"], + "review": packet["review"], "checks": packet["checks"], + "report": report, + } + finally: + store.close() + + def finish(self, run_id: str, disposition: str, reason: str) -> dict[str, Any]: + """Guided terminal host disposition over the existing fenced handoff gates.""" + if disposition not in {"accept", "reject", "revise"}: + raise ContractError("finish disposition is invalid") + if not isinstance(reason, str) or not reason.strip() or len(reason) > 2000: + raise ContractError("finish requires a non-empty reason of at most 2000 characters") + store = self._store() + try: + run, _, handoff = store.status_snapshot(run_id) + if (run["state"] != "awaiting_host" or run["phase"] is not None + or handoff is None or handoff.status != "open"): + raise ConflictError("run has no current open handoff to finish; inspect squad status RUN") + snapshot = self._review_snapshot(run) + if snapshot.get("task", {}).get("lead", {}).get("mode") != "host": + raise ConflictError("headless lead owns this handoff; run squad resume RUN") + packet = validate_saved_review_handoff(handoff.packet, snapshot) + for reference in packet["artifacts"]: + artifact = store.artifact_named(run_id, reference["name"]) + if (artifact is None or artifact["id"] != reference["artifact_id"] + or artifact["sha256"] != reference["sha256"]): + raise ConflictError("current handoff evidence reference is invalid") + try: + content = Path(artifact["path"]).read_bytes() + except OSError as exc: + raise ConflictError("current handoff evidence is missing") from exc + if (len(content) != artifact["byte_size"] + or hashlib.sha256(content).hexdigest() != reference["sha256"]): + raise ConflictError("current handoff evidence is corrupt") + body = { + "schema_version": 1, + "submission_id": f"terminal-{handoff.handoff_id}", + "disposition": disposition, + "reason": reason.strip(), + "evidence_refs": [ + {"artifact_id": ref["artifact_id"], "sha256": ref["sha256"]} + for ref in packet["artifacts"] + ], + } + decision = {**body, "submission_hash": request_hash(body)} + # Invalid decisions do not acquire a lease. Claim and completion + # still independently recheck their version and evidence fences. + self._review_gate(store, run_id, handoff, snapshot, decision) + version = run["version"] + finally: + store.close() + claimed = self.handoff_claim( + run_id, version, "terminal-operator", _initial_only=True, + ) + return self.handoff_complete(run_id, claimed["claim"], decision) + def events(self, run_id: str, after: int = 0, limit: int = 100) -> dict[str, Any]: store = self._store() try: return store.events_page(run_id, after, limit) @@ -1387,6 +1484,8 @@ def handoff_claim( expected_version: int, owner: str, prior_claim: dict[str, Any] | None = None, + *, + _initial_only: bool = False, ) -> dict[str, Any]: if type(expected_version) is not int or expected_version < 1: raise ContractError("handoff expected version is invalid") @@ -1405,7 +1504,9 @@ def handoff_claim( raise ConflictError( "headless review does not accept a host claim" ) - claim = store.claim_handoff(run_id, expected_version, owner, decoded) + claim = store.claim_handoff( + run_id, expected_version, owner, decoded, initial_only=_initial_only, + ) snapshot = store.handoff_snapshot(run_id) if snapshot is None: # Defensive: claim_handoff just verified it. raise ConflictError("claimed handoff is missing") diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index bd065f3..c8fe127 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -248,6 +248,15 @@ def _project(self, common_dir: Path) -> str: self.connection.execute("INSERT INTO projects(id, git_common_dir, created_at) VALUES(?,?,?)", (project_id, key, _utc_now())) return project_id + def runs_for_project(self, project: Path) -> list[dict[str, Any]]: + """Bounded choices from this canonical Git project, without registering it.""" + common_dir = git_common_dir(project) + return [dict(row) for row in self.connection.execute( + "SELECT r.id AS run_id,r.state,r.phase FROM runs r " + "JOIN projects p ON p.id=r.project_id WHERE p.git_common_dir=? " + "ORDER BY r.id LIMIT 21", (str(common_dir),), + ).fetchall()] + def claim_start(self, worktree: Path, idempotency_key: str, submitted_request: Any, owner_id: str, *, objective_outcome: bool = False) -> StartClaim: if not idempotency_key or not owner_id: raise ContractError("idempotency key and owner are required") @@ -4442,6 +4451,7 @@ def claim_handoff( prior_claim: HandoffClaim | None = None, *, now: datetime | None = None, + initial_only: bool = False, ) -> HandoffClaim: if (type(expected_version) is not int or expected_version < 1 or not isinstance(owner_id, str) or not owner_id): @@ -4467,6 +4477,9 @@ def claim_handoff( if (row["state"] != "awaiting_host" or row["phase"] is not None or row["status"] != "open" or row["version"] != expected_version): raise ConflictError("handoff is not claimable at that run version") + if (initial_only and row["kind"] == "host" + and row["claim_handoff_id"] == row["handoff_id"]): + raise ConflictError("handoff already has a host claim; use its saved claim to complete or renew it") live = bool( row["kind"] == "host" and row["active"] and row["claim_handoff_id"] == row["handoff_id"] diff --git a/plugin/core/src/devsquad/task_entry.py b/plugin/core/src/devsquad/task_entry.py index 43083c3..55cbccd 100644 --- a/plugin/core/src/devsquad/task_entry.py +++ b/plugin/core/src/devsquad/task_entry.py @@ -22,6 +22,7 @@ request, ) from .contracts import ContractError +from .probe_process import capture_probe_identity, close_probe, subscription_environment from .store import canonical_json from .validation import validate_profile, validate_task @@ -71,16 +72,6 @@ def _resolve_commit(repo: Path, reference: str) -> str: return _git(repo, "rev-parse", "--verify", f"{reference}^{{commit}}") -def _stop(process: subprocess.Popen[str]) -> None: - if process.poll() is None: - process.terminate() - try: - process.wait(timeout=3) - except subprocess.TimeoutExpired: - process.kill() - process.wait(timeout=3) - - def discover_codex_identity( repo: Path, *, @@ -102,6 +93,7 @@ def discover_codex_identity( process = subprocess.Popen( [binary, "app-server", "--listen", "stdio://"], cwd=repo, + env=subscription_environment(), stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, @@ -111,7 +103,10 @@ def discover_codex_identity( ) except OSError as exc: raise ContractError("Codex native process could not start") from exc + start_identity = capture_probe_identity(process) try: + if start_identity is None: + raise ContractError("Codex native process ownership is unavailable") assert process.stdin is not None and process.stdout is not None peer = JsonLinePeer(process.stdout, process.stdin) peer.send(initialize_request(1)) @@ -160,7 +155,7 @@ def read_native(request_id, method, params=None): except (EOFError, OSError, TimeoutError) as exc: raise ContractError("Codex model discovery did not complete") from exc finally: - _stop(process) + close_probe(process, start_identity=start_identity) candidates = [ model for model in models if model["supported_efforts"] @@ -417,6 +412,8 @@ def _detect_tests(repo: Path, oid: str) -> list[tuple[str, ...]]: "python3", "scripts/run-core-tests.py", )) + if _is_regular_blob(repo, oid, "scripts/generate-core-reference.py"): + detected.append(("python3", "scripts/generate-core-reference.py", "--check")) return detected @@ -441,9 +438,13 @@ def _checks( "timeout_seconds": min(timeout_seconds, 120), "required_to_pass": required, }] - seen: set[tuple[str, ...]] = set() - detected_ids = ("detected-tests", "detected-core-tests") - for check_id, argv in zip(detected_ids, _detect_tests(repo, target_oid)): + seen: set[tuple[str, ...]] = {tuple(checks[0]["argv"])} + for argv in _detect_tests(repo, target_oid): + check_id = ( + "generated-core-reference" if argv == ("python3", "scripts/generate-core-reference.py", "--check") + else "detected-core-tests" if argv[-1] == "scripts/run-core-tests.py" + else "detected-tests" + ) checks.append({ "id": check_id, "argv": list(argv), @@ -625,4 +626,5 @@ def build_managed_task( ), "scope": task["scope"], "checks": [check["id"] for check in task["checks"]], + "check_plan": task["checks"], } diff --git a/test/core/fakes/claude_delivery_cli.py b/test/core/fakes/claude_delivery_cli.py new file mode 100755 index 0000000..5f820ca --- /dev/null +++ b/test/core/fakes/claude_delivery_cli.py @@ -0,0 +1,35 @@ +#!/usr/bin/env python3 +"""Offline Claude stream fixture: repair only the installed-usability seed.""" + +import json +import os +from pathlib import Path +import sys + + +if "--version" in sys.argv: + print("2.1.220 (Claude Code)") + raise SystemExit(0) +if sys.argv[1:3] == ["auth", "status"]: + print(json.dumps({"loggedIn": True, "authMethod": "claude.ai", "apiProvider": "firstParty"})) + raise SystemExit(0) +if os.environ.get("DEVSQUAD_WORKER") != "1" or "--print" not in sys.argv or "--" not in sys.argv: + raise SystemExit("fixture accepts only a fenced offline writer invocation") + +path = Path("src/app.py") +seed = "def add(a, b):\n return a - b\n" +if path.is_symlink() or not path.is_file() or path.read_text() != seed: + raise SystemExit("fixture writer requires the exact seeded defect") +path.write_text("def add(a, b):\n return a + b\n") +model, session = "claude-sonnet-4-6", "offline-installed-writer-session" +records = [{ + "type": "assistant", "session_id": session, "parent_tool_use_id": None, + "message": {"role": "assistant", "model": model}, +}, { + "type": "result", "subtype": "success", "is_error": False, + "session_id": session, "result": "Corrected the bounded addition defect.", + "usage": {"input_tokens": 12, "output_tokens": 7}, + "modelUsage": {model: {"inputTokens": 12, "outputTokens": 7, "provider": "firstParty"}}, +}] +for record in records: + print(json.dumps(record), flush=True) diff --git a/test/core/fakes/codex_review_cli.py b/test/core/fakes/codex_review_cli.py index 0ba37b9..3e8db33 100755 --- a/test/core/fakes/codex_review_cli.py +++ b/test/core/fakes/codex_review_cli.py @@ -34,6 +34,17 @@ }), flush=True) elif method == "initialized": initialized = True + elif method == "account/read": + print(json.dumps({"id": request_id, "result": { + "account": {"type": "chatgpt", "id": "offline-fixture-account", "planType": "plus"}, + "requiresOpenaiAuth": True, + }}), flush=True) + elif method == "config/read": + print(json.dumps({"id": request_id, "result": { + "config": {"model_provider": "openai"}, + }}), flush=True) + elif method == "account/rateLimits/read": + print(json.dumps({"id": request_id, "result": {"rateLimits": None}}), flush=True) elif method == "model/list": if not initialized: print(json.dumps({ diff --git a/test/core/test_cli.py b/test/core/test_cli.py index aeeea52..a0b176b 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -33,7 +33,7 @@ def invoke(self, argv, service=None): stdout, stderr = io.StringIO(), io.StringIO() patcher = mock.patch.object(cli, "Service", return_value=service) if service is not None else contextlib.nullcontext() with patcher, contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): - code = cli.main(argv) + code = cli.main(argv if "--json" in argv else [*argv, "--json"]) lines = stdout.getvalue().splitlines() self.assertEqual(len(lines), 1, stdout.getvalue()) return code, json.loads(lines[0]), stderr.getvalue() @@ -266,6 +266,35 @@ def test_capacity_observe_dispatches_file(self): self.assert_success_envelope(payload, response) service.capacity_observe.assert_called_once_with(observation) + def test_finish_preserves_the_json_envelope_and_forwards_a_guided_choice(self): + for disposition in ("accept", "reject", "revise"): + with self.subTest(disposition=disposition): + service = mock.Mock() + response = {"run_id": "run-1", "state": "succeeded", "disposition": disposition} + service.finish.return_value = response + code, payload, stderr = self.invoke([ + "finish", "run-1", f"--{disposition}", "--reason", "Exact evidence assessed.", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.finish.assert_called_once_with("run-1", disposition, "Exact evidence assessed.") + service.resolve_run_id.assert_not_called() + + def test_normal_doctor_default_is_readable_and_keeps_unknown_proof_unknown(self): + report = {"core_version": "fixture", "ready": False, "adapters": [{ + "adapter": "claude", "status": "supported", "version": "fixture", + "authentication": {"status": "unauthenticated", "next_action": "claude auth login"}, + "operation_verified": None, + }], "supported_workflows": {"issue-delivery": {"supported": True, "ready": False}}} + output = io.StringIO() + with mock.patch.object(cli, "build_doctor_report", return_value=report), contextlib.redirect_stdout(output): + code = cli.main(["doctor", "--project-dir", str(self.root)]) + self.assertEqual(code, 1) + self.assertIn("authentication unauthenticated; operation unknown", output.getvalue()) + self.assertIn("claude auth login", output.getvalue()) + self.assertIn("issue-delivery: needs attention", output.getvalue()) + def test_outcome_add_and_report_dispatch(self): outcome = {"schema_version": 1, "outcome_id": "outcome-1"} outcome_file = self.root / "outcome.json" @@ -462,7 +491,9 @@ def test_handoff_claim_renew_and_complete_dispatch_parsed_objects(self): self.assertEqual(payload["error"]["code"], "INPUT_INVALID") def test_parser_and_json_file_failures_are_input_errors(self): - code, payload, _ = self.invoke(["status"]) + service = mock.Mock() + service.resolve_run_id.side_effect = ContractError("No saved runs for this Git project") + code, payload, _ = self.invoke(["status", "--runtime-dir", str(self.runtime)], service) self.assertEqual(code, 64) self.assertEqual(payload["error"]["code"], "INPUT_INVALID") diff --git a/test/core/test_diagnostics.py b/test/core/test_diagnostics.py new file mode 100644 index 0000000..ad9e67c --- /dev/null +++ b/test/core/test_diagnostics.py @@ -0,0 +1,442 @@ +import contextlib +import io +import json +import os +from pathlib import Path +import signal +import sys +import tempfile +import time +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad import cli, diagnostics, mcp_server, probe_process +from devsquad.contracts import ContractError +from devsquad.supervisor import _live_group_exists, process_start_identity + + +class DoctorReadinessTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-doctor-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.home = self.root / "home" + self.home.mkdir() + self.project = self.root / "project" + self.project.mkdir() + self.core = self.root / "core" + self.log = self.root / "probes.jsonl" + self.manager = mock.Mock( + mcp_sdk_available=True, mcp_sdk_supported=True, + mcp_sdk_version="2.2.0", squad_executable=ROOT / "plugin/core/bin/squad", + launcher_error=None, + ) + self.manager.inspect.side_effect = lambda template: { + "id": template.id, "installed": True, "ready": True, "status": "matching", + } + self.core_patch = mock.patch.object(diagnostics, "CORE_ROOT", self.core) + self.core_patch.start() + self.addCleanup(self.core_patch.stop) + + def install(self, name, *, version=None, response=None, returncode=0, raw=None, delay=0): + versions = { + "codex": "codex-cli 0.159.2", "claude": "2.1.220 (Claude Code)", + "grok": "1.0.46", "antigravity": "1.2.14", + } + version = version or versions[name] + binary = self.root / (name + "-fixture") + configuration = { + "name": name, "version": version, "response": response, + "returncode": returncode, "raw": raw, "log": str(self.log), "delay": delay, + } + binary.write_text("#!" + sys.executable + "\n" + """ +import json, os, sys, time +configuration = CONFIGURATION +def record(value): + with open(configuration['log'], 'a') as output: + output.write(json.dumps(value) + '\\n') +record({'argv': sys.argv[1:], 'environment': dict(os.environ)}) +if sys.argv[1:] == ['--version']: + print(configuration['version']) +elif sys.argv[1:] == ['auth', 'status', '--json']: + time.sleep(configuration['delay']) + print(configuration['raw'] if configuration['raw'] is not None else json.dumps(configuration['response'])) + print('private-stderr-token', file=sys.stderr) + sys.exit(configuration['returncode']) +elif sys.argv[1:] == ['app-server', '--listen', 'stdio://']: + for line in sys.stdin: + message = json.loads(line) + record({'request': message}) + if message['method'] == 'initialize': + print(json.dumps({'id': message['id'], 'result': {}}), flush=True) + elif message['method'] == 'account/read': + time.sleep(configuration['delay']) + if configuration['raw'] is not None: + print(configuration['raw'], flush=True) + else: + print(json.dumps({'id': message['id'], 'result': configuration['response']}), flush=True) +else: + sys.exit(99) +""".replace("CONFIGURATION", repr(configuration))) + binary.chmod(0o755) + directory = self.core / "adapters" / name + directory.mkdir(parents=True, exist_ok=True) + (directory / "adapter.json").write_text(json.dumps({ + "schema_version": 1, "name": name, + "transport": "native_protocol" if name == "codex" else "cli_exec", + "binary_candidates": [str(binary)], "capabilities": {}, + "verified_harness_versions": ( + [versions[name]] if name in {"codex", "claude"} else [] + ), + })) + return binary + + def report(self): + return diagnostics.build_doctor_report( + project=self.project, home=self.home, manager=self.manager, + ) + + def row(self, report, name): + return next(row for row in report["adapters"] if row["adapter"] == name) + + def probes(self): + return [json.loads(line) for line in self.log.read_text().splitlines()] + + def codex(self, response=None): + return self.install("codex", response=response or { + "account": {"type": "chatgpt", "email": "private@example.test", + "accountId": "private-account", "accessToken": "private-token"}, + "requiresOpenaiAuth": True, + }) + + def test_logged_out_claude_is_visible_and_delivery_is_unavailable(self): + self.codex() + self.install("claude", response={"loggedIn": False, "authMethod": "none"}, returncode=1) + report = self.report() + claude = self.row(report, "claude") + self.assertTrue(claude["installed"]) + self.assertTrue(claude["supported"]) + self.assertIs(claude["authenticated"], False) + self.assertFalse(claude["ready"]) + self.assertEqual(claude["authentication"]["next_action"], "claude auth login") + self.assertTrue(report["supported_workflows"]["branch-review"]["ready"]) + self.assertFalse(report["supported_workflows"]["issue-delivery"]["ready"]) + self.assertTrue(report["ready"]) + + def test_unverified_cli_never_passes_adapter_readiness_or_auth_probe(self): + self.install("codex", version="codex-cli 99.0.0") + report = self.report() + row = self.row(report, "codex") + self.assertEqual(row["status"], "unverified") + self.assertTrue(row["installed"]) + self.assertFalse(row["supported"]) + self.assertIsNone(row["authenticated"]) + self.assertFalse(report["adapter_ready"]) + self.assertFalse(report["ready"]) + self.assertEqual([row["argv"] for row in self.probes()], [["--version"]]) + + def test_unsupported_authentication_is_unknown_even_with_four_registrations(self): + self.install("grok") + self.install("antigravity") + report = self.report() + self.assertEqual(len(report["local_apps"]), 4) + self.assertTrue(report["local_app_access"]["ready"]) + self.assertTrue(all(row["registered"] for row in report["local_apps"])) + self.assertTrue(all(row["operation_verified"] is None for row in report["local_apps"])) + self.assertFalse(report["ready"]) + for row in report["adapters"]: + self.assertIsNone(row["authenticated"]) + self.assertIsNone(row["operation_verified"]) + self.assertFalse(row["supported"]) + self.assertTrue(all(probe["argv"] == ["--version"] for probe in self.probes())) + + def test_authentication_and_registration_do_not_invent_operation_proof(self): + self.codex() + self.install("claude", response={ + "loggedIn": True, "authMethod": "claude.ai", "email": "private@example.test", + "organizationId": "private-org", "apiKey": "private-key", + }) + report = self.report() + self.assertTrue(report["ready"]) + self.assertTrue(report["supported_workflows"]["issue-delivery"]["ready"]) + self.assertFalse(report["supported_workflows"]["council"]["ready"]) + for row in report["adapters"]: + self.assertIs(row["authenticated"], True) + self.assertIsNone(row["operation_verified"]) + output = json.dumps(report) + for secret in ("private@example.test", "private-account", "private-token", + "private-org", "private-key", "private-stderr-token"): + self.assertNotIn(secret, output) + + def test_claude_malformed_banner_missing_or_non_boolean_auth_stays_unknown(self): + cases = [ + "Welcome private-token\n" + json.dumps({"loggedIn": True, "authMethod": "claude.ai"}), + "not-json private-token", "[]", "{}", + json.dumps({"loggedIn": "true", "authMethod": "claude.ai"}), + json.dumps({"loggedIn": 1, "authMethod": "claude.ai"}), + json.dumps({"loggedIn": True, "authMethod": "private-token"}), + json.dumps({"loggedIn": True, "authMethod": None}), + json.dumps({"loggedIn": True, "authMethod": []}), + json.dumps({"loggedIn": True, "authMethod": "none"}), + '{"loggedIn":false,"loggedIn":true,"authMethod":"claude.ai"}', + ] + for raw in cases: + with self.subTest(raw=raw): + self.install("claude", raw=raw) + report = self.report() + self.assertIsNone(self.row(report, "claude")["authenticated"]) + self.assertFalse(report["ready"]) + self.assertNotIn("private-token", json.dumps(report)) + + def test_api_key_auth_does_not_make_subscription_workflows_available(self): + self.install("claude", response={"loggedIn": True, "authMethod": "api_key"}) + self.codex({"account": {"type": "apiKey", "apiKey": "private-key"}, + "requiresOpenaiAuth": True}) + report = self.report() + self.assertFalse(report["ready"]) + self.assertFalse(report["adapter_ready"]) + for row in report["adapters"]: + self.assertTrue(row["authenticated"]) + self.assertFalse(row["authentication"]["subscription_supported"]) + self.assertFalse(row["ready"]) + self.assertNotIn("private-key", json.dumps(report)) + + def test_codex_null_account_and_unknown_provider_state_are_not_ready(self): + for response, authenticated in ( + ({"account": None, "requiresOpenaiAuth": True}, False), + ({"account": None, "requiresOpenaiAuth": False}, None), + ({"account": {"type": "unknown", "token": "private-token"}, "requiresOpenaiAuth": True}, None), + ({"account": {"type": []}, "requiresOpenaiAuth": True}, None), + ({"account": {"type": "chatgpt"}, "requiresOpenaiAuth": "true"}, None), + ({"requiresOpenaiAuth": True}, None), + ): + with self.subTest(response=response): + self.install("codex", response=response) + report = self.report() + self.assertIs(self.row(report, "codex")["authenticated"], authenticated) + self.assertFalse(report["ready"]) + self.assertNotIn("private-token", json.dumps(report)) + + def test_codex_protocol_is_nongenerating_and_environment_drops_credentials(self): + self.codex() + with mock.patch.dict(os.environ, { + "USER": "fixture-user", "OPENAI_API_KEY": "private-key", + "ANTHROPIC_API_KEY": "private-key", "CODEX_HOME": "/private/override", + "CLAUDE_CODE_OAUTH_TOKEN": "private-token", "PRIVATE_PROVIDER_OVERRIDE": "secret", + }): + report = self.report() + self.assertTrue(report["ready"]) + probes = self.probes() + requests = [probe["request"] for probe in probes if "request" in probe] + self.assertEqual([request["method"] for request in requests], + ["initialize", "initialized", "account/read"]) + self.assertEqual(requests[-1]["params"], {"refreshToken": False}) + for probe in (probe for probe in probes if "environment" in probe): + environment = probe["environment"] + self.assertEqual(environment["HOME"], str(self.home)) + self.assertEqual(environment["USER"], "fixture-user") + self.assertIn("PATH", environment) + # Python and macOS may add these locale values at process startup. + self.assertLessEqual(set(environment), { + "HOME", "USER", "PATH", "LC_CTYPE", "__CF_USER_TEXT_ENCODING", + }) + + def test_oversized_output_and_version_banners_are_redacted_before_parse(self): + for name in ("claude", "codex"): + with self.subTest(name=name): + self.install(name, raw="private-token" + ("x" * diagnostics.MAX_PROBE_BYTES)) + report = self.report() + self.assertIsNone(self.row(report, name)["authenticated"]) + self.assertNotIn("private-token", json.dumps(report)) + for name in ("claude", "codex"): + with self.subTest(name=name, depth="excessive"): + self.install(name, raw="[" * 1500 + "0" + "]" * 1500) + self.assertIsNone(self.row(self.report(), name)["authenticated"]) + self.install("claude", version="login failed private-token") + report = self.report() + self.assertIsNone(self.row(report, "claude")["version"]) + self.assertNotIn("private-token", json.dumps(report)) + + def test_codex_malformed_banner_error_or_uncorrelated_reply_stays_unknown(self): + for raw in ( + "Welcome private-token", "[]", "{}", + json.dumps({"id": 2, "result": None}), + json.dumps({"id": 99, "result": {"account": {"type": "chatgpt"}, "requiresOpenaiAuth": True}}), + json.dumps({"id": 2, "error": {"message": "private-token"}}), + ): + with self.subTest(raw=raw): + self.install("codex", raw=raw) + report = self.report() + self.assertIsNone(self.row(report, "codex")["authenticated"]) + self.assertFalse(report["ready"]) + self.assertNotIn("private-token", json.dumps(report)) + + def test_auth_checks_time_out_and_reap_the_owned_process(self): + start_process = diagnostics.subprocess.Popen + for name in ("claude", "codex"): + with self.subTest(name=name): + self.install(name, delay=60) + processes = [] + def capture_process(*args, **kwargs): + process = start_process(*args, **kwargs) + processes.append(process) + return process + with ( + mock.patch.object(diagnostics, "PROBE_TIMEOUT_SECONDS", 0.5), + mock.patch.object(diagnostics, "AUTH_TIMEOUT_SECONDS", 0.5), + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=capture_process) as spawned, + ): + report = self.report() + self.assertIsNone(self.row(report, name)["authenticated"]) + self.assertFalse(self.row(report, name)["ready"]) + for call in spawned.call_args_list: + self.assertEqual(set(call.kwargs["env"]), {"HOME", "USER", "PATH"}) + for process in processes: + self.assertIsNotNone(process.poll()) + self.assertTrue(process.stdout.closed) + + def test_exited_probe_parent_cannot_leave_term_ignoring_descendant_alive(self): + child_record = self.root / "probe-child.json" + code = """ +import json, os, signal, sys, time +from pathlib import Path +sys.path.insert(0, SOURCE) +from devsquad.supervisor import process_start_identity +read_ready, write_ready = os.pipe() +if os.fork() == 0: + os.close(read_ready) + signal.signal(signal.SIGTERM, signal.SIG_IGN) + Path(RECORD).write_text(json.dumps({'pid': os.getpid(), 'start': process_start_identity(os.getpid())})) + os.write(write_ready, b'r') + os.close(write_ready) + time.sleep(60) + os._exit(0) +os.close(write_ready) +os.read(read_ready, 1) +os.close(read_ready) +print('codex-cli 0.159.2', flush=True) +os._exit(0) +""".replace("SOURCE", repr(str(ROOT / "plugin/core/src"))).replace("RECORD", repr(str(child_record))) + processes = [] + start_process = diagnostics.subprocess.Popen + def capture_process(*args, **kwargs): + process = start_process(*args, **kwargs) + processes.append(process) + return process + started = time.monotonic() + try: + with ( + mock.patch.object(diagnostics, "PROBE_TIMEOUT_SECONDS", 0.5), + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=capture_process), + self.assertRaises(TimeoutError), + ): + diagnostics._probe_output( + [sys.executable, "-c", code], project=self.project, + environment=diagnostics._environment(self.home), + ) + self.assertLess(time.monotonic() - started, 3) + self.assertTrue(child_record.exists(), "descendant readiness barrier was not reached") + process = processes[0] + self.assertEqual(process.returncode, 0) + self.assertFalse(_live_group_exists(process.pid), "owned descendant survived probe cleanup") + self.assertTrue(process.stdout.closed) + finally: + if processes: + process = processes[0] + child = json.loads(child_record.read_text()) if child_record.exists() else None + if child and process_start_identity(child["pid"]) == child["start"]: + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + process.wait(timeout=2) + deadline = time.monotonic() + 2 + while _live_group_exists(process.pid) and time.monotonic() < deadline: + time.sleep(0.02) + self.assertFalse(_live_group_exists(process.pid), "controlled fixture cleanup failed") + + def test_reused_probe_identity_is_not_safe_to_signal(self): + process = mock.Mock(pid=987654, returncode=0, stderr=None) + with ( + mock.patch.object(probe_process, "_probe_group_exists", return_value=True), + mock.patch.object(probe_process, "process_start_identity", return_value="new-start"), + mock.patch.object(probe_process.os, "killpg") as signal_group, + self.assertRaisesRegex(ContractError, "identity changed"), + ): + probe_process.close_probe(process, start_identity="original-start") + signal_group.assert_not_called() + process.stdout.close.assert_called_once() + + def test_missing_group_is_confirmed_before_reap_without_signaling_stale_pid(self): + process = mock.Mock(pid=987654, returncode=0, stderr=None) + with ( + mock.patch.object(probe_process, "_probe_group_exists", return_value=False), + mock.patch.object(probe_process, "process_start_identity", return_value="new-start"), + mock.patch.object(probe_process.os, "killpg") as signal_group, + ): + probe_process.close_probe(process, start_identity="original-start") + signal_group.assert_not_called() + process.wait.assert_called_once() + + def test_identity_change_after_term_blocks_kill_escalation(self): + process = mock.Mock(pid=987654, returncode=None, stderr=None) + identity = {"value": "original-start"} + def record_signal(pgid, value): + identity["value"] = "reused-start" + with ( + mock.patch.object(probe_process, "_probe_group_exists", return_value=True), + mock.patch.object(probe_process, "process_start_identity", side_effect=lambda pid: identity["value"]), + mock.patch.object(probe_process.os, "killpg", side_effect=record_signal) as signal_group, + self.assertRaisesRegex(ContractError, "identity changed"), + ): + probe_process.close_probe(process, start_identity="original-start") + signal_group.assert_called_once_with(process.pid, signal.SIGTERM) + + def test_missing_cli_never_claims_authentication_or_support(self): + self.codex().unlink() + report = self.report() + row = self.row(report, "codex") + self.assertEqual(row["status"], "unavailable") + self.assertFalse(row["installed"]) + self.assertFalse(row["supported"]) + self.assertIsNone(row["authenticated"]) + self.assertIsNone(row["operation_verified"]) + self.assertFalse(report["ready"]) + self.assertFalse(self.log.exists()) + + def test_cli_and_mcp_doctor_keep_the_same_machine_envelope(self): + self.codex() + report = self.report() + stdout = io.StringIO() + with mock.patch.object(cli, "build_doctor_report", return_value=report): + with contextlib.redirect_stdout(stdout): + code = cli.main(["doctor", "--json"]) + self.assertEqual(code, 0) + envelope = json.loads(stdout.getvalue()) + self.assertEqual(envelope["schema_version"], 1) + self.assertTrue(envelope["ok"]) + self.assertEqual(envelope["data"], report) + bridge = mcp_server.MCPBridge(self.root / "runtime", mock.Mock(), environment={}) + with mock.patch.object(mcp_server, "build_doctor_report", return_value=report): + self.assertEqual(bridge.doctor(), envelope) + + def test_read_only_report_keeps_config_and_ledger_unchanged(self): + self.codex() + configuration = self.home / ".claude.json" + configuration.write_text('{"secret":"private-token","keep":true}') + database = self.project / "state.sqlite" + database.write_bytes(b"unchanged-ledger") + before = {str(path): path.read_bytes() for directory in (self.home, self.project) + for path in directory.rglob("*") if path.is_file()} + self.report() + after = {str(path): path.read_bytes() for directory in (self.home, self.project) + for path in directory.rglob("*") if path.is_file()} + self.assertEqual(before, after) + self.manager.setup.assert_not_called() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_installed_usability.py b/test/core/test_installed_usability.py new file mode 100644 index 0000000..ce72bf6 --- /dev/null +++ b/test/core/test_installed_usability.py @@ -0,0 +1,182 @@ +"""Fresh installed normal commands, with only provider binaries replaced.""" + +import hashlib +import json +import os +from pathlib import Path +import re +import shutil +import subprocess +import sys +import tempfile +import time +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +import test_cli as cli_fixtures + + +class InstalledNormalUsabilityTest(unittest.TestCase): + def setUp(self): + python = cli_fixtures.InstalledWheelMigrationTest.build_python() + if python is None: + self.skipTest("offline supported build interpreter is unavailable") + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-installed-ux-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.home = self.root / "home" + self.home.mkdir() + self.provider_bin = self.root / "providers" + self.provider_bin.mkdir() + for name, target in ( + ("codex", ROOT / "test/core/fakes/codex_review_cli.py"), + ("claude", ROOT / "test/core/fakes/claude_delivery_cli.py"), + ("python3", Path(python)), + ("git", Path(shutil.which("git"))), + ): + (self.provider_bin / name).symlink_to(target) + codex_home = self.home / ".codex" + codex_home.mkdir() + (codex_home / "auth.json").write_text("{}\n") + (codex_home / "auth.json").chmod(0o600) + claude_home = self.home / ".claude" + claude_home.mkdir() + (claude_home / ".credentials.json").write_text("{}\n") + (claude_home / ".credentials.json").chmod(0o600) + self.runtime = self.root / "runtime" + self.environment = { + "HOME": str(self.home), "USER": "devsquad-offline-test", + "PATH": f"{self.provider_bin}{os.pathsep}/usr/bin{os.pathsep}/bin", + "CODEX_HOME": str(codex_home), "PYTHONWARNINGS": "error::ResourceWarning", + "PYTHONDONTWRITEBYTECODE": "1", + "PIP_NO_INDEX": "1", "PIP_DISABLE_PIP_VERSION_CHECK": "1", + "DEVSQUAD_PYTHON": python, "DEVSQUAD_RUNTIME_DIR": str(self.runtime), + "DEVSQUAD_INSTALL_ROOT": str(self.root / "installed"), + "DEVSQUAD_BIN_DIR": str(self.root / "bin"), + } + installed = subprocess.run( + ["/bin/bash", str(ROOT / "scripts/install-core.sh"), "--source-core", str(ROOT / "plugin/core"), "--json"], + cwd=self.root, env=self.environment, text=True, capture_output=True, timeout=60, + ) + self.assertEqual((installed.returncode, installed.stderr), (0, ""), installed.stdout) + self.installation = json.loads(installed.stdout) + self.assertTrue(self.installation["changed"]) + self.assertFalse(self.installation["installed"]["mcp"]) + self.launcher = self.root / "bin/squad" + self.run_ids = set() + self.addCleanup(self.cancel_unfinished_runs) + + def git(self, repo, *arguments): + return subprocess.run([str(self.provider_bin / "git"), "-C", str(repo), *arguments], + env=self.environment, text=True, capture_output=True, check=True).stdout + + def project(self, name, *, defective): + repo = self.root / name + repo.mkdir() + self.git(repo, "init", "-qb", "main") + self.git(repo, "config", "user.name", "Offline Test") + self.git(repo, "config", "user.email", "test@example.invalid") + (repo / "src").mkdir() + (repo / "tests").mkdir() + (repo / "src/app.py").write_text(f"def add(a, b):\n return a {'-' if defective else '+'} b\n") + (repo / "tests/test_app.py").write_text( + "import unittest\nfrom src.app import add\n" + "class AdditionTest(unittest.TestCase):\n" + " def test_addition(self):\n self.assertEqual(add(2, 3), 5)\n" + ) + self.git(repo, "add", ".") + self.git(repo, "commit", "-qm", "seeded fixture") + return repo + + def command(self, repo, *arguments, expected=0): + result = subprocess.run([str(self.launcher), *arguments], cwd=repo, env=self.environment, + text=True, capture_output=True, timeout=60) + self.run_ids.update(re.findall(r"^Run: ([a-f0-9-]{36})$", result.stdout, re.MULTILINE)) + evidence = result.stdout + if result.returncode != expected: + for run_id in re.findall(r"^Run: ([a-f0-9-]{36})$", result.stdout, re.MULTILINE): + report = subprocess.run([str(self.launcher), "result", run_id, "--json"], cwd=repo, + env=self.environment, text=True, capture_output=True, timeout=15) + if report.returncode == 0: + for artifact in json.loads(report.stdout)["data"]["artifacts"]: + if artifact["name"] == "receipt.json": + evidence += "\n" + Path(artifact["path"]).read_text() + self.assertEqual((result.returncode, result.stderr), (expected, ""), evidence) + return result.stdout + + def receipt(self, repo, run_id): + result = json.loads(self.command(repo, "result", run_id, "--json"))["data"] + self.assertTrue(result["ready"]) + self.assertEqual(result["state"], "succeeded") + for artifact in result["artifacts"]: + self.assertEqual(hashlib.sha256(Path(artifact["path"]).read_bytes()).hexdigest(), artifact["sha256"]) + reference = next(item for item in result["artifacts"] if item["name"] == "receipt.json") + return json.loads(Path(reference["path"]).read_text()) + + def cancel_unfinished_runs(self): + for run_id in self.run_ids: + subprocess.run([str(self.launcher), "cancel", run_id, "--json"], cwd=self.root, + env=self.environment, text=True, capture_output=True, timeout=15) + deadline = time.monotonic() + 10 + while time.monotonic() < deadline: + result = subprocess.run([str(self.launcher), "status", run_id, "--json"], cwd=self.root, + env=self.environment, text=True, capture_output=True, timeout=5) + if result.returncode == 0 and json.loads(result.stdout)["data"]["state"] in {"succeeded", "failed", "cancelled"}: + break + time.sleep(0.05) + else: + self.fail(f"installed fixture cleanup did not terminalize {run_id}") + + def test_fresh_installed_review_and_fix_finish_using_normal_commands(self): + review_repo = self.project("review-project", defective=False) + output = self.command(review_repo, "status", expected=64) + self.assertIn("No saved runs", output) + output = self.command(self.root, "review", "--project-dir", str(review_repo), "--base", "main", "--wait", expected=2) + self.assertIn("Review: clean", output) + self.assertIn("squad finish", output) + review_id = json.loads(self.command(review_repo, "status", "--json"))["data"]["run_id"] + self.command(review_repo, "finish", "--accept", "--reason", "Inspected the exact saved review.") + self.assertIn("receipt.md", self.command(review_repo, "result")) + review_receipt = self.receipt(review_repo, review_id) + self.assertEqual(review_receipt["accounting"]["worker_invocations"], 1) + + fix_repo = self.project("fix-project", defective=True) + self.assertIn("No saved runs", self.command(fix_repo, "status", expected=64)) + original_oid = self.git(fix_repo, "rev-parse", "HEAD") + original_source = (fix_repo / "src/app.py").read_bytes() + seeded_failure = subprocess.run([str(self.provider_bin / "python3"), "-m", "unittest", "discover", "-s", "tests"], + cwd=fix_repo, env=self.environment, capture_output=True, text=True) + self.assertEqual(seeded_failure.returncode, 1) + output = self.command(self.root, "fix", "Correct add so it returns the sum.", "--project-dir", str(fix_repo), "--write-path", "src/app.py", "--wait", expected=2) + self.assertIn("Check detected-tests: passed", output) + fix_id = json.loads(self.command(fix_repo, "status", "--json"))["data"]["run_id"] + self.assertNotEqual(fix_id, review_id) + self.command(fix_repo, "finish", "--accept", "--reason", "Seeded addition test and independent review pass.") + receipt = self.receipt(fix_repo, fix_id) + self.assertEqual(receipt["accounting"]["worker_invocations"], 2) + identities = [item["observed_identity"] for item in receipt["attempts"]] + self.assertEqual([item["harness"] for item in identities], ["claude", "codex"]) + self.assertNotEqual(identities[0]["model_id"], identities[1]["model_id"]) + self.assertEqual(identities[0]["verification_scope"], "reported_model") + self.assertEqual(identities[0]["native_evidence"]["writer_messages"]["message_count"], 1) + self.assertTrue(all(item["required_to_pass"] and item["status"] == "passed" for item in receipt["checks"])) + self.assertTrue(all(item["integrity"]["status"] == "verified" for item in receipt["checks"])) + self.assertEqual(self.git(fix_repo, "rev-parse", "HEAD"), original_oid) + self.assertEqual((fix_repo / "src/app.py").read_bytes(), original_source) + self.assertEqual(self.git(fix_repo, "status", "--porcelain"), "") + + # Omitted IDs stay project-scoped even when newer unrelated runs exist. + self.assertEqual(json.loads(self.command(review_repo, "status", "--json"))["data"]["run_id"], review_id) + self.command(review_repo, "review", "--base", "main", "--idempotency-key", "installed-second-review", "--wait", expected=2) + output = self.command(review_repo, "status", expected=75) + self.assertIn("Multiple saved runs", output) + self.assertIn(review_id, output) + self.assertNotIn(fix_id, output) + output = self.command(review_repo, "result", expected=75) + self.assertIn("specify RUN", output) + self.assertTrue(json.loads(self.command(review_repo, "result", review_id, "--json"))["data"]["ready"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index dccf290..3d9d555 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -413,7 +413,11 @@ def test_installed_app_drift_controls_readiness_but_unavailable_apps_do_not(self ] manager.inspect.side_effect = rows templates = (mock.Mock(id="codex"), mock.Mock(id="claude-code")) - adapters = [{"adapter": "codex", "status": "supported"}] + adapters = [{ + "adapter": "codex", "status": "supported", "installed": True, + "supported": True, "authenticated": True, "ready": True, + "operation_verified": None, + }] with ( mock.patch.object(diagnostics, "_adapter_rows", return_value=adapters), mock.patch.object(diagnostics, "load_integrations", return_value=templates), diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py index 6cd8392..5f7c59f 100644 --- a/test/core/test_task_entry.py +++ b/test/core/test_task_entry.py @@ -4,6 +4,7 @@ import json import os from pathlib import Path +import signal import subprocess import sys import tempfile @@ -642,6 +643,37 @@ def test_check_discovery_combines_bash_and_core_runner(self): ], ) + def test_generated_reference_gate_follows_committed_regular_target_and_deduplicates(self): + repo = self._init_repo() + for directory in ("test/core", "plugin/core/src/devsquad", "scripts"): + (repo / directory).mkdir(parents=True) + (repo / "test/run.sh").write_text("#!/usr/bin/env bash\nexit 0\n") + (repo / "test/core/test_sample.py").write_text("# fixture test\n") + (repo / "plugin/core/src/devsquad/__init__.py").write_text("") + (repo / "scripts/run-core-tests.py").write_text("# fixture runner\n") + generator = repo / "scripts/generate-core-reference.py" + generator.write_text("# fixture generator\n") + target = self._commit(repo, "required project gates") + generator.unlink() + generator.symlink_to("run-core-tests.py") + symlink_target = self._commit(repo, "generator becomes a symlink") + for oid, expected in ((target, 1), (symlink_target, 0)): + with self.subTest(target=oid): + task, _ = build_managed_task( + workflow="issue-delivery", project_dir=repo, base_ref=oid, + target_ref=oid, goal="Keep generated contracts current.", + codex_identity=self.codex, + checks=parse_checks(["python3 scripts/generate-core-reference.py --check"] if expected else []), + ) + references = [check for check in task["checks"] if check["argv"] == [ + "python3", "scripts/generate-core-reference.py", "--check", + ]] + self.assertEqual(len(references), expected) + self.assertTrue(all(check["required_to_pass"] for check in task["checks"])) + self.assertIn("detected-core-tests", [check["id"] for check in task["checks"]]) + if expected: + self.assertEqual(references[0]["id"], "generated-core-reference") + def test_check_discovery_deduplicates_supplied_argv_matching_detected(self): repo = self._init_repo() (repo / "test").mkdir() @@ -730,7 +762,10 @@ def test_codex_discovery_selects_requested_exact_model_and_effort(self): with ( mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), - mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process), + mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process) as spawn, + mock.patch("devsquad.task_entry.capture_probe_identity", return_value="fixture-native-start") as capture, + mock.patch("devsquad.task_entry.close_probe") as close, + mock.patch.dict(os.environ, {"HOME": self.temp.name, "USER": "offline-test", "PATH": "/fixture/bin", "OPENAI_API_KEY": "synthetic-denied", "ANTHROPIC_API_KEY": "synthetic-denied", "CODEX_HOME": "/fixture/alternate-home"}), mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), mock.patch("devsquad.task_entry.receive_response", return_value={"result": {}}), mock.patch("devsquad.task_entry.discover_models", return_value=[]), @@ -748,13 +783,17 @@ def test_codex_discovery_selects_requested_exact_model_and_effort(self): "model_family": "gpt-6", "effort": "xhigh", }) - process.terminate.assert_called_once_with() - process.wait.assert_called_once_with(timeout=3) + capture.assert_called_once_with(process) + close.assert_called_once_with(process, start_identity="fixture-native-start") + self.assertEqual(spawn.call_args.kwargs["env"], {"HOME": self.temp.name, "USER": "offline-test", "PATH": "/fixture/bin"}) + self.assertTrue(spawn.call_args.kwargs["start_new_session"]) with ( mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process), + mock.patch("devsquad.task_entry.capture_probe_identity", return_value="fixture-native-start"), + mock.patch("devsquad.task_entry.close_probe"), mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), mock.patch("devsquad.task_entry.receive_response", return_value={"result": {}}), mock.patch("devsquad.task_entry.discover_models", return_value=[]), @@ -767,6 +806,83 @@ def test_codex_discovery_selects_requested_exact_model_and_effort(self): requested_effort="ultra", ) + def test_native_discovery_does_not_read_protocol_without_owned_identity(self): + manifest = mock.Mock(verified_versions=("codex-cli fixture",)) + manifest.resolve_binary.return_value = "/fixture/codex" + process = mock.Mock(stdin=io.StringIO(), stdout=io.StringIO()) + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", return_value=process), + mock.patch("devsquad.task_entry.capture_probe_identity", return_value=None), + mock.patch("devsquad.task_entry.close_probe") as close, + mock.patch("devsquad.task_entry.JsonLinePeer") as peer, + ): + with self.assertRaisesRegex(ContractError, "ownership is unavailable"): + discover_codex_identity(self.repo) + peer.assert_not_called() + close.assert_called_once_with(process, start_identity=None) + + def test_native_discovery_cleans_owned_child_after_provider_parent_exits(self): + ready = Path(self.temp.name) / "native-child.json" + child_code = ( + "import json,os,signal,sys,time\n" + "signal.signal(signal.SIGTERM, signal.SIG_IGN)\n" + "with open(sys.argv[1], 'w') as handle: json.dump({'pid':os.getpid()},handle)\n" + "while True: time.sleep(1)\n" + ) + provider_code = ( + "import json,os,subprocess,sys,time\n" + f"subprocess.Popen([sys.executable, '-c', {child_code!r}, {str(ready)!r}], stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n" + f"while not os.path.exists({str(ready)!r}): time.sleep(.01)\n" + "for line in sys.stdin:\n" + " item=json.loads(line)\n" + " if item.get('method') == 'initialize':\n" + " print(json.dumps({'id':item['id'],'result':{}}),flush=True)\n" + " elif item.get('method') == 'model/list':\n" + " print(json.dumps({'id':item['id'],'result':{'data':[{'id':'gpt-fixture','supportedReasoningEfforts':[{'reasoningEffort':'low'}]}],'nextCursor':None}}),flush=True)\n" + " os._exit(0)\n" + ) + manifest = mock.Mock(verified_versions=("codex-cli fixture",)) + manifest.resolve_binary.return_value = "/fixture/codex" + real_popen = subprocess.Popen + spawned = {} + def native_popen(argv, *positional, **keywords): + if argv[:2] != ["/fixture/codex", "app-server"]: + return real_popen(argv, *positional, **keywords) + process = real_popen([sys.executable, "-c", provider_code], *positional, **keywords) + spawned["process"] = process + return process + from devsquad.task_entry import discover_models as real_discover_models + def discover_then_confirm_parent_exit(*arguments, **keywords): + models = real_discover_models(*arguments, **keywords) + self.assertEqual(spawned["process"].wait(timeout=2), 0) + return models + try: + with ( + mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), + mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), + mock.patch("devsquad.task_entry.subprocess.Popen", side_effect=native_popen), + mock.patch("devsquad.task_entry.discover_models", side_effect=discover_then_confirm_parent_exit), + ): + identity = discover_codex_identity(self.repo) + self.assertEqual(identity["model_id"], "gpt-fixture") + child = json.loads(ready.read_text())["pid"] + observed = subprocess.run(["/bin/ps", "-p", str(child), "-o", "stat="], + text=True, capture_output=True, timeout=1).stdout.strip() + self.assertTrue(not observed or observed.startswith("Z"), f"owned child {child} survives: {observed}") + finally: + process = spawned.get("process") + if process is not None: + try: + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + process.wait(timeout=2) + for stream in (process.stdin, process.stdout, process.stderr): + if stream is not None: + stream.close() + def test_normal_native_discovery_caches_across_projects_and_ingests_shared_quota(self): from datetime import datetime, timedelta, timezone manifest = mock.Mock(verified_versions=("codex-cli fixture",)) @@ -795,6 +911,8 @@ def reply(peer, request_id, **unused): mock.patch("devsquad.task_entry.AdapterManifest.load", return_value=manifest), mock.patch("devsquad.task_entry.harness_version", return_value="codex-cli fixture"), mock.patch("devsquad.task_entry.subprocess.Popen", side_effect=native_popen), + mock.patch("devsquad.task_entry.capture_probe_identity", return_value="fixture-native-start"), + mock.patch("devsquad.task_entry.close_probe"), mock.patch("devsquad.task_entry.JsonLinePeer", return_value=mock.Mock()), mock.patch("devsquad.task_entry.receive_response", side_effect=reply), mock.patch("devsquad.task_entry.discover_models", return_value=[{"id": "gpt-fixture", "supportedReasoningEfforts": ["low"]}]) as discovery, diff --git a/test/core/test_terminal_ux.py b/test/core/test_terminal_ux.py new file mode 100644 index 0000000..f910f82 --- /dev/null +++ b/test/core/test_terminal_ux.py @@ -0,0 +1,282 @@ +"""Normal terminal operations use real offline workers and the saved gates.""" + +import contextlib +import copy +import io +import json +import os +from datetime import datetime, timedelta, timezone +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest import mock + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) + +from devsquad import cli +from devsquad import detached +from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.store import ConflictError, Store +from devsquad.task_entry import build_managed_task + + +class TerminalUxTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-terminal-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.repo = self.new_repo("project") + self.service = Service(self.root / "runtime") + self.identity = { + "harness": "codex", "harness_version": "codex-cli fixture", + "model_id": "gpt-fixture", "model_family": "gpt", "effort": "low", + } + + def new_repo(self, name): + repo = self.root / name + repo.mkdir() + subprocess.run(["git", "init", "-qb", "main", str(repo)], check=True) + subprocess.run(["git", "-C", str(repo), "config", "user.name", "Test"], check=True) + subprocess.run(["git", "-C", str(repo), "config", "user.email", "test@example.invalid"], check=True) + (repo / "app.py").write_text("VALUE = 1\n") + subprocess.run(["git", "-C", str(repo), "add", "."], check=True) + subprocess.run(["git", "-C", str(repo), "commit", "-qm", "fixture"], check=True) + return repo + + def invoke(self, argv): + output, errors = io.StringIO(), io.StringIO() + with contextlib.redirect_stdout(output), contextlib.redirect_stderr(errors): + code = cli.main([*argv, "--runtime-dir", str(self.service.runtime)]) + return code, output.getvalue(), errors.getvalue() + + def normal_run(self, key="normal-review", workflow="review", check=None): + original_start = self.service.start + def offline_start(task, *args): + fixtures = {"_internal_review_fixture": { + "verdict": "clean", "summary": "Exact fixture candidate is clean.", "findings": [], + }} + if task["workflow"] == "issue-delivery": + fixtures["_internal_implementation_fixture"] = { + "writes": [{"path": "app.py", "content": "VALUE = 2\n"}], "delay_seconds": 0, + } + return original_start(task, *args, **fixtures) + argv = [workflow] + if workflow == "fix": + argv += ["Correct the bounded value.", "--write-path", "app.py"] + argv += ["--base", "HEAD", "--project-dir", str(self.repo), "--idempotency-key", key, "--wait", "--json"] + if check: + argv += ["--check", check] + with mock.patch.object(cli, "discover_codex_identity", return_value=self.identity), mock.patch.object(cli, "_service", return_value=self.service), mock.patch.object(self.service, "start", side_effect=offline_start): + code, output, errors = self.invoke(argv) + if code != 2: + run = self.service.resolve_run_id(None, self.repo) + result = self.service.result(run) + output += "\n" + next(Path(item["path"]).read_text() for item in result["artifacts"] if item["name"] == "receipt.json") + self.assertEqual((code, errors), (2, ""), output) + data = json.loads(output)["data"] + self.assertEqual(data["state"], "awaiting_host") + return data["run_id"] + + def test_zero_one_multiple_and_unrelated_project_resolution(self): + with self.assertRaisesRegex(ContractError, "No saved runs.*squad review"): + self.service.resolve_run_id(None, self.repo) + code, output, _ = self.invoke(["status", "--project-dir", str(self.repo)]) + self.assertEqual(code, 64) + self.assertIn("No saved runs", output) + unrelated = self.new_repo("unrelated") + store = Store(self.service.database, self.service.artifacts) + try: + other = store.claim_start(unrelated, "other", {}, "fixture").run_id + finally: + store.close() + with self.assertRaisesRegex(ContractError, "No saved runs"): + self.service.resolve_run_id(None, self.repo) + run = self.normal_run() + self.assertEqual(self.service.resolve_run_id(None, self.repo), run) + linked = self.root / "linked" + subprocess.run(["git", "-C", str(self.repo), "worktree", "add", "-q", "--detach", str(linked), "HEAD"], check=True) + self.assertEqual(self.service.resolve_run_id(None, linked), run) + self.assertEqual(self.service.resolve_run_id(run, unrelated), run) + second = self.normal_run("second") + with self.assertRaises(ConflictError) as raised: + self.service.resolve_run_id(None, self.repo) + self.assertIn(run, str(raised.exception)) + self.assertIn(second, str(raised.exception)) + self.assertNotIn(other, str(raised.exception)) + code, output, _ = self.invoke(["result", "--project-dir", str(self.repo)]) + self.assertEqual(code, 75) + self.assertIn(run, output) + self.assertIn(second, output) + self.assertNotIn(other, output) + + def test_normal_review_and_fix_finish_without_task_or_decision_json(self): + for workflow in ("review", "fix"): + with self.subTest(workflow=workflow): + run = self.normal_run(workflow, workflow) + code, output, errors = self.invoke(["finish", run, "--accept", "--reason", "Reviewed exact evidence."]) + self.assertEqual((code, errors), (0, ""), output) + self.assertIn("succeeded", output) + self.assertIn(f"squad result {run}", output) + result = self.service.result(run) + self.assertTrue(result["ready"]) + self.assertEqual(result["state"], "succeeded") + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "No terminal replay.") + self.assertEqual((self.repo / "app.py").read_text(), "VALUE = 1\n") + + def test_guided_finish_does_not_take_over_or_accept_failed_required_check(self): + run = self.normal_run("failed-check", "fix", "python3 -c 'raise SystemExit(7)'") + with self.assertRaises(ContractError): + self.service.finish(run, "accept", "Cannot override objective failure.") + self.assertEqual(self.service.status(run)["state"], "awaiting_host") + self.service.handoff_claim(run, self.service.status(run)["version"], "other-owner") + for advance in (0, 700): + with mock.patch("devsquad.store._authoritative_now", return_value=datetime.now(timezone.utc) + timedelta(seconds=advance)): + with self.assertRaisesRegex(ConflictError, "claim"): + self.service.finish(run, "reject", "Another claim owns the packet.") + + def test_reject_and_revision_exhaustion_use_the_existing_terminal_gates(self): + rejected = self.normal_run("reject") + decision = self.service.finish(rejected, "reject", "Review remains incomplete.") + self.assertEqual((decision["state"], decision["disposition"]), ("failed", "reject")) + revised = self.normal_run("no-revision-budget") + decision = self.service.finish(revised, "revise", "Recheck the bounded review.") + self.assertEqual((decision["state"], decision["disposition"]), ("failed", "revise")) + self.assertTrue(self.service.result(revised)["ready"]) + + def test_finish_refuses_corrupt_current_evidence_before_claiming(self): + run = self.normal_run("corrupt-evidence") + before = self.service.status(run) + store = Store(self.service.database, self.service.artifacts) + try: + handoff = store.handoff_snapshot(run) + artifact = store.artifact_named(run, handoff.packet["artifacts"][0]["name"]) + finally: + store.close() + Path(artifact["path"]).write_text("corrupt fixture evidence\n") + with self.assertRaisesRegex(ConflictError, "corrupt"): + self.service.finish(run, "accept", "Must verify saved evidence.") + after = self.service.status(run) + self.assertEqual(after["version"], before["version"]) + self.assertIsNone(after["handoff"]["claimed_by"]) + + def test_status_result_omitted_ids_and_readable_default(self): + run = self.normal_run() + for command in ("status", "result"): + code, output, errors = self.invoke([command, "--project-dir", str(self.repo)]) + self.assertEqual((code, errors), (0, ""), output) + self.assertIn(run, output) + self.assertFalse(output.startswith("{")) + code, output, _ = self.invoke(["status", "--project-dir", str(self.repo), "--json"]) + self.assertEqual(code, 0) + self.assertEqual(json.loads(output)["data"]["run_id"], run) + + def test_native_offline_normal_review_uses_real_discovery_and_public_finish(self): + fake_bin = self.root / "native-bin" + fake_bin.mkdir() + (fake_bin / "codex").symlink_to(ROOT / "test/core/fakes/codex_review_cli.py") + fake_home = self.root / "native-home" + fake_home.mkdir() + (fake_home / "auth.json").write_text("{}\n") + (fake_home / "auth.json").chmod(0o600) + with mock.patch.dict(os.environ, { + "PATH": f"{fake_bin}{os.pathsep}{os.environ.get('PATH', '')}", + "CODEX_HOME": str(fake_home), + }): + code, output, errors = self.invoke([ + "review", "--base", "HEAD", "--project-dir", str(self.repo), + "--idempotency-key", "native-normal", "--wait", + ]) + self.assertEqual((code, errors), (2, ""), output) + self.assertIn("Review: clean", output) + self.assertIn("Check candidate-diff-check: passed", output) + self.assertIn("handoff.md", output) + run = self.service.resolve_run_id(None, self.repo) + code, output, errors = self.invoke([ + "finish", "--project-dir", str(self.repo), "--accept", + "--reason", "Reviewed exact offline native fixture evidence.", "--json", + ]) + self.assertEqual((code, errors), (0, ""), output) + self.assertEqual(json.loads(output)["data"]["run_id"], run) + self.assertTrue(self.service.result(run)["ready"]) + + def test_headless_wait_continues_awaiting_host_and_queued_handoff(self): + service = mock.Mock() + service.status.side_effect = [ + {"run_id": "headless", "state": "awaiting_host", "version": 8, "next_action": "continue_headless_lead"}, + {"run_id": "headless", "state": "queued", "version": 9, "next_action": None}, + {"run_id": "headless", "state": "succeeded", "version": 12}, + ] + with mock.patch.object(cli.time, "sleep"): + response, code = cli._wait_for_run(service, {"run_id": "headless"}) + self.assertEqual((code, response["data"]["state"]), (0, "succeeded")) + service.resume.assert_called_once_with("headless") + + def test_headless_wait_keeps_observing_after_detached_continuation_race(self): + service = mock.Mock() + service.status.side_effect = [ + {"run_id": "headless", "state": "awaiting_host", "version": 8, "next_action": "continue_headless_lead"}, + {"run_id": "headless", "state": "running", "version": 10}, + {"run_id": "headless", "state": "succeeded", "version": 12}, + ] + service.resume.side_effect = ConflictError("detached owner already continued") + with mock.patch.object(cli.time, "sleep"): + response, code = cli._wait_for_run(service, {"run_id": "headless"}) + self.assertEqual((code, response["data"]["state"]), (0, "succeeded")) + + def test_queued_headless_handoff_is_not_a_candidate_review(self): + task, _ = build_managed_task( + workflow="issue-delivery", project_dir=self.repo, base_ref="HEAD", + target_ref="HEAD", goal="Correct the bounded fixture value.", + codex_identity=self.identity, write_paths=("app.py",), + ) + lead = copy.deepcopy(task["routing"]["profiles"]["profiles"][-1]) + lead.update({"id": "fixture-lead", "model_id": "gpt-fixture-lead"}) + task["routing"]["profiles"]["profiles"].append(lead) + task["routing"]["policy"]["roles"]["lead"] = [{"kind": "profile", "id": lead["id"]}] + task["lead"] = {"mode": "headless"} + with mock.patch.object(self.service, "_spawn_daemon"): + started = self.service.start( + task, "headless-status", + _internal_implementation_fixture={"writes": [{"path": "app.py", "content": "VALUE = 2\n"}], "delay_seconds": 0}, + _internal_review_fixture={"verdict": "clean", "summary": "Exact candidate is clean.", "findings": []}, + _internal_lead_fixture={"disposition": "accept", "reason": "Exact evidence is sufficient."}, + ) + run_id = started["run_id"] + def run_stage(): + store = Store(self.service.database, self.service.artifacts) + try: + run = store.run(run_id) + finally: + store.close() + with mock.patch.dict(os.environ, {"PYTHONPATH": run["package_path"]}): + self.assertEqual(detached.main([ + "--database", str(self.service.database), "--artifacts", str(self.service.artifacts), + "--run-id", run_id, "--expected-version", str(run["version"]), + "--package-digest", run["package_digest"], + ]), 0) + run_stage() + with mock.patch.object(self.service, "_spawn_daemon"): + self.service.resume(run_id) + with mock.patch.object(Service, "resume"): + run_stage() + waiting = self.service.status(run_id) + self.assertEqual(waiting["next_action"], "continue_headless_lead") + with self.assertRaisesRegex(ConflictError, "headless"): + self.service.finish(run_id, "accept", "Host must not replace the selected lead.") + with mock.patch.object(self.service, "_spawn_daemon"): + self.service.resume(run_id) + queued = self.service.status(run_id) + self.assertEqual(queued["state"], "queued") + self.assertEqual(queued["handoff"]["status"], "open") + self.assertIsNone(queued["next_action"]) + run_stage() + self.assertEqual(self.service.status(run_id)["state"], "succeeded") + + +if __name__ == "__main__": + unittest.main() From 752514659969139f61bccf4fac48e6e2a0fad062 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 15:49:26 -0700 Subject: [PATCH 181/197] WIP checkpoint: docs: scope Antigravity acceptance to agy CLI (2026-10-02 15:49) --- docs/RUNTIME-GUIDE.md | 7 ++++--- docs/plans/engineering-team/IMPLEMENTATION.md | 2 +- docs/plans/engineering-team/backlog.json | 2 +- 3 files changed, 6 insertions(+), 5 deletions(-) diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index 7861695..b794ca5 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -242,7 +242,7 @@ Implemented surfaces and evidence: | Terminal | Standalone install and real saved-run cancellation; fresh installed normal review/fix/finish flow verified with offline provider binaries | | Codex App/CLI | Matching MCP registration and a fresh installed-runtime `squad_status` receipt on 0.155.0-alpha.9.2 | | Claude Code local Code tab | Matching registration and real Claude MCP handoff; local Code-tab UI proof remains open | -| Antigravity local IDE/CLI | Matching registration and a live Gemini `squad_status` receipt with one project-scoped grant | +| Antigravity CLI (`agy`) | Matching registration and a live Gemini `squad_status` receipt with one project-scoped grant; IDE is outside the clarified request | | Grok Build | Matching registration and real Grok 1.0.46 MCP status operation | The historical installed surface evidence source is @@ -252,8 +252,9 @@ The installed normal-entry evidence is Later verified runtime proofs, including the accepted two-model delivery, actual Claude handoff, Grok MCP and Gemini CLI/MCP recheck, are recorded in [`R8-installed-workflows-2026-10-01.json`](plans/engineering-team/evidence/R8-installed-workflows-2026-10-01.json). -These are operation-scoped receipts; desktop UI proofs and the remaining R6/R8 -acceptance gates remain open. Doctor separates installed binaries, supported +These are operation-scoped receipts; the Claude local Code-tab UI proof and +remaining R6/R8 acceptance gates remain open. Antigravity acceptance uses the +CLI, not IDE trust or UI. Doctor separates installed binaries, supported adapter versions, non-generating authentication checks, registrations and operation verification. Unknown verification stays unknown. A CLI that is installed but unsupported or missing subscription authentication does not make diff --git a/docs/plans/engineering-team/IMPLEMENTATION.md b/docs/plans/engineering-team/IMPLEMENTATION.md index bd6d10a..e84cde9 100644 --- a/docs/plans/engineering-team/IMPLEMENTATION.md +++ b/docs/plans/engineering-team/IMPLEMENTATION.md @@ -137,7 +137,7 @@ Work in this order: 2. Keep legacy plugin mode available during migration. Reconcile hook registration only when duplicate evidence exists; September review found no current duplicate. New hooks call the shared route source after its tests pass and remain fast/network-free. No Python dependency imposed on legacy mode. 3. Create generated command/schema examples and concise install/operate/recover guides. Mark implemented vs deferred features; link completion claims to receipts. Keep historical ADR/audit statements dated. 4. Add offline CI for legacy and core suites (Bash 3.2/macOS compatibility and chosen Python floor/current version), package-content checks, native protocol compatibility fixtures and optional MCP tests. Live provider/app tests remain explicit bounded smoke runs. Document supported protocol ranges, capability drift and the optional native Claude→Codex session-import path, while retaining portable artifact handoffs for every host. -5. Verify terminal CLI, Codex App, Claude Code App local Code tab, Antigravity local IDE/CLI and Grok Build against the same saved runtime. Check native capabilities/profile identity as used, not by brand inference. +5. Verify terminal CLI, Codex App, Claude Code App local Code tab, Antigravity CLI (`agy`) and Grok Build against the same saved runtime. The user's October 2 clarification excludes Antigravity IDE trust/UI from this acceptance. Check native capabilities/profile identity as used, not by brand inference. 6. If the classifier trial is retained, package it as an isolated optional extra with pinned model assets and explicit setup. Test absent extras, offline operation, unsupported hardware, cancellation and default-off equivalence. Hooks remain inference/download-free and network-free; neither a hosted key nor heavy model dependencies become a core-install requirement. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index b744ad2..d7a64fe 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -245,7 +245,7 @@ "availability": "portable_redacted" } ], - "blocker": "Claude authentication and actual CLI/MCP handoff now pass. Required desktop UI proofs remain unverified; Antigravity IDE control is permission-denied. Portable CLI proof is not substituted for desktop local Code-tab proof." + "blocker": "Claude authentication and actual CLI/MCP handoff now pass. Claude local Code-tab operation remains unverified. The user's October 2 clarification excludes Antigravity IDE: verify the updated agy CLI against the final accepted installation. Portable Claude CLI proof is not substituted for the local Code-tab proof." }, { "id": "M5", From 181a1a042186a748950ec042fc61ce45cfe5a336 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 15:55:04 -0700 Subject: [PATCH 182/197] WIP checkpoint: R6: checkpoint atomic guided recovery and independent expiry edge findings (2026-10-02 15:55) --- docs/RUNTIME-GUIDE.md | 9 +- docs/plans/engineering-team/RESUME.md | 40 +++- docs/plans/engineering-team/backlog.json | 4 +- ...terminal-readiness-partial-2026-10-02.json | 24 +- plugin/core/src/devsquad/cli.py | 19 +- plugin/core/src/devsquad/service.py | 4 + plugin/core/src/devsquad/store.py | 97 ++++++++- test/core/test_handoff_store.py | 3 +- test/core/test_terminal_ux.py | 205 +++++++++++++++++- 9 files changed, 376 insertions(+), 29 deletions(-) diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index b794ca5..13913fc 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -102,8 +102,13 @@ finish command. That pause exits 2; it is saved work awaiting your assessment. Use `--reject` or `--revise` instead of `--accept` when appropriate, with a reason. Finish binds every artifact from the exact current packet and applies the same claim, independent-review and required-check gates as the low-level -API. It refuses terminal replays and existing host claims; use the saved claim -for a handoff already owned by an app. The normal lead is the terminal host; +API. It refuses terminal replays and app-owned host claims; use the saved claim +for a handoff already owned by an app. If interrupted after acquiring its own +claim, finish saves the exact decision atomically with that claim. Status shows +the exact retry command. Only that same packet, disposition and reason can +recover the live claim or its expired fence; an owner name alone never permits +recovery. If the decision was already submitted, use `squad resume RUN_ID` to +finish the saved continuation. The normal lead is the terminal host; an explicitly configured headless run follows its existing lead through a temporary handoff when observed with `--wait`. Omitting `--wait` returns the run ID immediately. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 96d676e..9dd6254 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -94,14 +94,40 @@ repaired by the shared bounded helper; `03124c79` has 106 affected tests/ found a real P1: interruption after guided finish acquires its claim but before completion leaves no saved claim for retry; initial-only refuses it even after expiry. A controlled offline reproduction confirms this, not a usage timeout. -The terminal agent is repairing durable exact intent/claim recovery without -app-claim takeover or changed disposition. Integrate the repair, then launch -one frozen full gate and exact R6 audit. No native/full gate or production -installation change has started for this candidate. +Repair checkpoint `22a4ed8a` is now integrated: the canonical decision and exact +packet hash commit atomically with the claim. Live identical retries preserve +the claim; expired identical retries acquire a fresh fence only with the latest +matching durable terminal marker. App claims (including the same owner name), +changed decisions, corruption, cancellation and stale/new handoff races fail +closed. Saved submissions resume normally; Next commands are copyable. Agent +gates: 150 affected tests/180.750s and final 45 CLI/UX tests/48.316s; Bash 227, +reference and diff checks pass. Its first expanded gate exposed only an old +isolated bf3 schema-15 test expectation; root already expected 16. Historical +schema-four fixture checks are retained with the current supported-version +assertion. Root's integrated 61 terminal/CLI/store/handoff tests pass in +55.963s with ResourceWarning strict. Independent follow-up passed ten existing +repair tests/26.811s but found two edge cases: an `expired_claim` rejection +poisons identical retry even after its own fresh fence, and a reason beginning +with `--` makes the emitted separate `--reason` argument invalid. The terminal +agent owns the narrow follow-up: preserve the full rejected row canonically in +append-only audit history before a gated operational-row recovery; keep all +ordinary app replay rules unchanged. Attach the quoted reason with `--reason=`. +Integrate/review/test this follow-up before one frozen full suite and exact +native R6 audit. No full/native gate has started. Production remains accepted +R5/schema16. R7 Council remains in isolated `r7-council`; its controlled stage flow is -partial. Default-deny macOS own-evidence/peer-ledger denial and native binary -version probes are boundary mechanics, not a live Council receipt. No new -native Council generation has been made. Preserve all attached worktrees. +partial. Its source is checkpointed at `42979ba4` with 21 Council tests and +67 shared-contract tests (two optional SDK skips), Bash/reference/diff gates. +Default-deny macOS own-evidence/peer-ledger denial and native bootstrap/catalog +probes are boundary mechanics, not a live Council receipt. Native HTTPS, +reserved-launch cancellation, comparison and final acceptance remain open. +A controlled public-start/actual-worker scheduling-barrier reproduction proves +the immutable installed R5/schema16 client can acquire a Council headless +handoff that the new client rejects, stranding its completed lead. Both stores +were authoritatively schema16; temporary runs were cancelled and owned worker +PIDs confirmed absent. The Council agent owns a minimal schema17 compatibility +epoch and old-client/active-upgrade tests. No production edit or new native +Council generation has been made. Preserve all attached worktrees. Desktop control worked for scoped inspection. Claude's local Code tab selected only this DevSquad project on `codex/engineering-team`, with an empty prompt; no proof diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index d7a64fe..6efbba5 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -47,8 +47,8 @@ {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "complete", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-closure-2026-10-02.json", "checkpoint": "Accepted dfe9976/source equivalent f87060b and installed schema16. Initial native findings repaired; exact follow-up e046a174 clean/accepted. Full 504 tests/511.241s, two optional SDK skips/no errors/failures/unraisable; nine actual installed SDK tests pass/no skips. Public complete trial/evaluate/qualify/promote/new-run/held-out regression/rollback, shared call/active deadline and all terminal projection origins pass. Safe old-schema upgrade and idempotence/drift gates pass. Failed history retained; automatic experimentation and Jev stay off."}, - {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Readiness, guided finish, human defaults, project-scoped run choices, committed-target checks and actual temporary installed normal review/fix fixtures integrated; shared catalog descendant cleanup repaired and focused root gates pass. Independent review found stranded guided-finish claim after interruption; repair required before full/native acceptance and installed gates. Current production remains accepted R5."}, - {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Implementation isolated in attached r7-council worktree. Controlled offline proposal/critic/lead flow and macOS boundary probes are partial; no accepted Council integration, comparison or native generation receipt yet."}, + {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Readiness, guided finish, human defaults, project-scoped run choices, committed-target checks and actual temporary installed normal review/fix fixtures integrated; shared catalog descendant cleanup repaired. Independent interrupted-finish P1 is repaired by atomic exact intent/current claim recovery at 22a4ed8a; 150 affected plus final 45 CLI/UX agent tests pass. Root integrated/full/native/installed acceptance remains pending. Current production remains accepted R5."}, + {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Isolated partial source 42979ba4: 21 Council tests, 67 shared tests/two optional SDK skips and Bash/reference gates pass. Controlled actual-worker reproduction confirms old schema16 client can strand a headless Council handoff; minimal epoch17 repair and old-client/upgrade gates are required. Native HTTPS isolation, reserved-launch cancellation, predeclared comparison, integrated/full/native/installed acceptance remain open."}, {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} ], "milestones": [ diff --git a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json index 9b7b73f..250a74d 100644 --- a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json @@ -2,7 +2,7 @@ "schema_version": 1, "work_package": "R6", "recorded_on": "2026-10-02", - "status": "integrated_partial_repair_required", + "status": "recovery_repair_integrated_followup_required", "base_revision": "8f2c1c6a7d9edfe7cc3d78ecbc59ce105be636b2", "integrated_checkpoints": [ "dd709804e70e0791614319a9ae5c2597768d7318", @@ -10,12 +10,15 @@ "b785d04013042960b25b47292c3c3a5ead528c87", "d0eec55c9dbede205e3c8831b7c224d3d77114d3", "4a6bc15d81017aa996be9a21e6c81bc02ebd17a6", - "03124c79ba53cb1b12296bb554024605e4cadbc7" + "03124c79ba53cb1b12296bb554024605e4cadbc7", + "22a4ed8ac85b95861f4ba9725ca729388b7a9f49" ], "contract": "SOL-REVIEW-FOLLOWUP R6/G3/G4; unchanged low-level MCP envelopes, fenced host disposition, Bash 3.2 and optional jq", "implemented": [ "Readable normal terminal commands; --json retains the versioned automation envelope", - "Guided finish binds the exact current packet/artifact hashes and existing independent-review/check/claim fences; rejects replay and prior host claims", + "Guided finish binds the exact current packet/artifact hashes and existing independent-review/check/claim fences; rejects replay and app-owned prior host claims", + "Finish intent and claim commit atomically; only an identical canonical decision whose latest acquisition event matches the current packet/owner/fence/expiry recovers its live or expired guided claim. App claims with the same owner name and later renewals cannot reuse old terminal authority", + "Saved submissions continue through resume; human status prints a copyable, shell-quoted exact intent retry command with separate guidance", "Omitted IDs require exactly one canonical current-project run; zero and multiple choices are explicit; unrelated runs never selected", "Doctor distinguishes installed, registered, exact supported version, subscription authentication, workflow readiness and unknown operation proof", "Bounded non-generating Claude auth and Codex account/read with no API keys or alternate credential-home overrides", @@ -28,12 +31,15 @@ {"revision": "b785d040", "tests": 103, "seconds": 89.115, "skips": 0, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"}, {"revision": "d0eec55c", "tests": 1, "seconds": 11.955, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"}, {"revision": "4a6bc15d", "tests": 41, "seconds": 11.356, "optional_sdk_skips": 2, "bash_assertions": 227}, - {"revision": "03124c79", "tests": 106, "seconds": 95.408, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"} + {"revision": "03124c79", "tests": 106, "seconds": 95.408, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"}, + {"revision": "22a4ed8a", "tests": 150, "seconds": 180.750, "failures": 0, "errors": 0, "scope": "Affected repair gate before final command-copyability edit"}, + {"revision": "22a4ed8a", "tests": 45, "seconds": 48.316, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Final CLI/terminal repair checkpoint"} ], "root_affected_gates": [ {"command": "python3 -m unittest test_diagnostics test_terminal_ux test_installed_usability test_cli test_task_entry test_mcp test_handoff_store", "tests": 114, "seconds": 65.066, "optional_sdk_skips": 2, "failures": 0, "errors": 0, "scope": "Integrated first five checkpoints, before final catalog cleanup reuse"}, {"command": "python3 -m unittest test_delivery_workflow test_delivery_identity_runtime test_handoff_service", "tests": 32, "seconds": 66.877, "failures": 0, "errors": 0}, - {"command": "python3 -m unittest test_task_entry test_installed_usability", "tests": 27, "seconds": 25.389, "skips": 0, "failures": 0, "errors": 0, "scope": "All six integrated checkpoints including final catalog cleanup reuse"} + {"command": "python3 -m unittest test_task_entry test_installed_usability", "tests": 27, "seconds": 25.389, "skips": 0, "failures": 0, "errors": 0, "scope": "All six integrated checkpoints including final catalog cleanup reuse"}, + {"command": "python3 -m unittest test_terminal_ux test_cli test_handoff_store test_handoff_service", "tests": 61, "seconds": 55.963, "failures": 0, "errors": 0, "scope": "Final integrated interrupted-finish repair and current-schema assertion; ResourceWarning strict"} ], "failure_history": [ "Readiness red baseline: nine tests, three failures and 17 subtest errors before implementation", @@ -43,7 +49,11 @@ "Fresh installed fixture seeded check wrote original-checkout bytecode; PYTHONDONTWRITEBYTECODE=1 restored the declared pristine fixture contract without weakening source gates", "Root integrated affected command ran 116 tests in 69.606s, two optional SDK skips and two loader errors from nonexistent test_review_service/test_delivery_service module names; not a passing command gate and not a runtime defect", "A separate follow-up invocation mistakenly included nonexistent test_delivery_runtime; corrected existing-module/full gates are required", - "Independent bounded terminal review reproduced a P1: interruption after guided finish claim commit before completion strands the run without saved claim; same finish retry rejects both live and expired own claim. Durable exact-intent recovery is being repaired before full/native acceptance" + "Independent bounded terminal review reproduced a P1: interruption after guided finish claim commit before completion strands the run without saved claim; same finish retry rejects both live and expired own claim. Durable exact-intent recovery is integrated at 22a4ed8a; final root/full/native acceptance remains pending", + "Repair red reproduction: three selected tests had one failure and two errors before durable intent implementation", + "First expanded repair-agent gate failed a stale isolated bf3 current-schema expectation of 15; root already expected 16 after dfe9976. Changed only the current-schema assertion to SUPPORTED_SCHEMA_VERSION and retained the historical schema-four fixture checks; not a current root runtime failure", + "Independent 22a4ed8a follow-up: ten existing repair tests pass in 26.811s, but a lease expiring after claim acquisition and before submission creates a durable expired_claim rejection. A fresh own retry advances fence 2 to 3/version 16 to 17 yet the stable decision replays that old rejection indefinitely. Follow-up must preserve complete immutable rejection history while recovering only exact fresh guided authority", + "Independent 22a4ed8a command-copyability edge: saved reason --deferred is printed as --reason --deferred and argparse rejects it. Attach the shell-quoted reason to --reason= and test actual command round-trip" ], "fresh_install_scope": { "providers": "offline native CLI protocol fixtures only", @@ -65,7 +75,7 @@ "unrelated_servers_or_settings": "not changed" }, "remaining_gates": [ - "Guided finish interrupted claim/intent recovery repair and adversarial regressions", + "Expiry-rejection poisoning and leading-hyphen reason follow-up, final integrated regressions and independent repair review", "Frozen full gate", "Independent exact R6 review and accepted immutable checkpoint", "Safe installed update, non-generating live readiness, idempotence/drift and actual installed SDK tests", diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index daefbfb..2790077 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -658,10 +658,13 @@ def _next_command(data: dict[str, Any]) -> str: run_id = data.get("run_id", "RUN") action = data.get("next_action") if action == "claim_handoff": + pending = (data.get("handoff_view") or {}).get("pending_finish") + if pending: + return f"squad finish {run_id} --{pending['disposition']} --reason {shlex.quote(pending['reason'])}" owner = (data.get("handoff") or {}).get("claimed_by") if owner: return f"complete or renew the saved claim in {owner}; this handoff already has an owner" - return f'squad finish {run_id} --accept --reason "your assessment of the saved evidence" (or --reject / --revise)' + return f'squad finish {run_id} --accept --reason "your assessment of the saved evidence"' if action in {"continue_headless_lead", "resume_candidate_review", "handoff_submission_saved"}: return f"squad resume {run_id}" if action == "recovery_file_required": @@ -673,6 +676,16 @@ def _next_command(data: dict[str, Any]) -> str: return f"squad status {run_id}" +def _next_lines(data: dict[str, Any]) -> list[str]: + lines = [f"Next: {_next_command(data)}"] + if data.get("next_action") == "claim_handoff": + if (data.get("handoff_view") or {}).get("pending_finish"): + lines.append("Guidance: retry the exact saved intent; disposition and reason must match.") + elif not (data.get("handoff") or {}).get("claimed_by"): + lines.append("Guidance: use --reject or --revise instead of --accept if the evidence requires it.") + return lines + + def _handoff_lines(data: dict[str, Any]) -> list[str]: view = data.get("handoff_view") if not view: @@ -734,7 +747,7 @@ def _human_response(command: str, response: dict[str, Any]) -> str: lines.append(f"Checks: {', '.join(data.get('checks', []))}") next_data = data.get("service") or data lines.extend(_handoff_lines(next_data)) - lines.append(f"Next: {_next_command(next_data)}") + lines.extend(_next_lines(next_data)) else: run_id = data.get("run_id", "unknown") lines.append(f"Run {run_id}: {data.get('state', 'unknown')}") @@ -753,7 +766,7 @@ def _human_response(command: str, response: dict[str, Any]) -> str: if command == "cancel" and data.get("state") == "cancelling": lines.append("Cancellation is saved; worker cleanup is still running.") if command != "result" or not data.get("ready"): - lines.append(f"Next: {_next_command(data)}") + lines.extend(_next_lines(data)) return "\n".join(lines) diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 287f7a7..57a4905 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -1360,6 +1360,7 @@ def handoff_view(self, run_id: str) -> dict[str, Any]: "candidate_sha256": packet["candidate_sha256"], "review": packet["review"], "checks": packet["checks"], "report": report, + "pending_finish": store.terminal_finish_decision(run_id, handoff.handoff_id), } finally: store.close() @@ -1411,6 +1412,7 @@ def finish(self, run_id: str, disposition: str, reason: str) -> dict[str, Any]: store.close() claimed = self.handoff_claim( run_id, version, "terminal-operator", _initial_only=True, + _terminal_decision=decision, ) return self.handoff_complete(run_id, claimed["claim"], decision) @@ -1486,6 +1488,7 @@ def handoff_claim( prior_claim: dict[str, Any] | None = None, *, _initial_only: bool = False, + _terminal_decision: dict[str, Any] | None = None, ) -> dict[str, Any]: if type(expected_version) is not int or expected_version < 1: raise ContractError("handoff expected version is invalid") @@ -1506,6 +1509,7 @@ def handoff_claim( ) claim = store.claim_handoff( run_id, expected_version, owner, decoded, initial_only=_initial_only, + terminal_decision=_terminal_decision, ) snapshot = store.handoff_snapshot(run_id) if snapshot is None: # Defensive: claim_handoff just verified it. diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index c8fe127..70e417e 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -4443,6 +4443,64 @@ def recorded_handoff_submission( return entry return None + def _terminal_finish_for_claim( + self, run_id: str, row: sqlite3.Row, + ) -> dict[str, Any] | None: + """Only the exact latest acquisition event proves terminal authority.""" + if (row["kind"] != "host" or not row["active"] + or row["owner_id"] != "terminal-operator" + or row["claim_handoff_id"] != row["handoff_id"]): + return None + event = self.connection.execute( + "SELECT type,payload FROM events WHERE run_id=? " + "AND type IN ('handoff.acquired','handoff.taken_over','handoff.renewed') " + "ORDER BY id DESC LIMIT 1", (run_id,), + ).fetchone() + if event is None: + return None + try: + payload = json.loads(event["payload"]) + if not isinstance(payload, dict) or "terminal_finish" not in payload: + return None + if (set(payload) != {"action", "expires_at", "fencing_token", "handoff_id", "owner_id", "terminal_finish"} + or canonical_json(payload) != event["payload"]): + raise ConflictError("persisted terminal finish marker is invalid") + marker = payload["terminal_finish"] + if (not isinstance(marker, dict) + or set(marker) != {"schema_version", "packet_sha256", "decision"} + or type(marker["schema_version"]) is not int + or marker["schema_version"] != 1): + raise ConflictError("persisted terminal finish intent is invalid") + # A later app claim or renewal invalidates older terminal authority, + # including when it happens to use the same human-readable owner. + if (type(payload["fencing_token"]) is not int + or payload["fencing_token"] != row["fencing_token"] + or payload["expires_at"] != row["lease_expires_at"] + or payload["handoff_id"] != row["handoff_id"] + or payload["owner_id"] != row["owner_id"] + or event["type"] != f"handoff.{payload['action']}" + or marker["packet_sha256"] != row["packet_sha256"]): + return None + self._validated_handoff_decision(run_id, marker["decision"]) + if marker["decision"]["submission_id"] != f"terminal-{row['handoff_id']}": + raise ConflictError("persisted terminal finish targets a different handoff") + return marker["decision"] + except (TypeError, ValueError, ContractError) as exc: + raise ConflictError("persisted terminal finish intent is invalid") from exc + + def terminal_finish_decision( + self, run_id: str, handoff_id: str, + ) -> dict[str, Any] | None: + """Read a pending guided choice without exposing or inventing a claim.""" + row = self.connection.execute( + "SELECT h.id AS handoff_id,h.packet_sha256,c.kind,c.owner_id," + "c.fencing_token,c.active,c.handoff_id AS claim_handoff_id,c.lease_expires_at " + "FROM runs r JOIN handoffs h ON h.run_id=r.id JOIN claims c ON c.run_id=r.id " + "WHERE r.id=? AND h.id=? AND r.state='awaiting_host' AND r.phase IS NULL " + "AND h.status='open'", (run_id, handoff_id), + ).fetchone() + return self._terminal_finish_for_claim(run_id, row) if row is not None else None + def claim_handoff( self, run_id: str, @@ -4452,19 +4510,24 @@ def claim_handoff( *, now: datetime | None = None, initial_only: bool = False, + terminal_decision: dict[str, Any] | None = None, ) -> HandoffClaim: if (type(expected_version) is not int or expected_version < 1 or not isinstance(owner_id, str) or not owner_id): raise ContractError("handoff claim requires a run version and owner") if prior_claim is not None and not isinstance(prior_claim, HandoffClaim): raise ContractError("prior handoff claim is invalid") + if terminal_decision is not None and ( + not initial_only or prior_claim is not None + or owner_id != "terminal-operator"): + raise ContractError("terminal finish requires its own initial guided claim") self.connection.execute("BEGIN IMMEDIATE") try: current = _authoritative_now(now) timestamp = current.isoformat() expires_at = (current + timedelta(seconds=HOST_LEASE_SECONDS)).isoformat() row = self.connection.execute( - "SELECT r.state,r.phase,r.version,h.id AS handoff_id,h.status," + "SELECT r.state,r.phase,r.version,h.id AS handoff_id,h.status,h.packet_sha256," "c.kind,c.owner_id,c.fencing_token,c.active,c.handoff_id AS claim_handoff_id," "c.lease_expires_at " "FROM runs r JOIN handoffs h ON h.run_id=r.id " @@ -4477,15 +4540,39 @@ def claim_handoff( if (row["state"] != "awaiting_host" or row["phase"] is not None or row["status"] != "open" or row["version"] != expected_version): raise ConflictError("handoff is not claimable at that run version") - if (initial_only and row["kind"] == "host" - and row["claim_handoff_id"] == row["handoff_id"]): - raise ConflictError("handoff already has a host claim; use its saved claim to complete or renew it") + terminal_marker = None + if terminal_decision is not None: + self._validated_handoff_decision(run_id, terminal_decision) + if terminal_decision["submission_id"] != f"terminal-{row['handoff_id']}": + raise ConflictError("terminal finish decision targets a different handoff") + if (not terminal_decision["reason"].strip() + or len(terminal_decision["reason"]) > 2000): + raise ContractError("terminal finish requires a bounded non-empty reason") + terminal_marker = { + "schema_version": 1, "packet_sha256": row["packet_sha256"], + "decision": json.loads(canonical_json(terminal_decision)), + } + prior_host_claim = row["kind"] == "host" and row["claim_handoff_id"] == row["handoff_id"] + if initial_only and prior_host_claim: + saved = self._terminal_finish_for_claim(run_id, row) if terminal_marker is not None else None + if saved is None: + raise ConflictError("handoff already has a host claim; use its saved claim to complete or renew it") + if canonical_json(saved) != canonical_json(terminal_decision): + raise ConflictError("a different terminal finish intent is pending; retry its exact saved decision") live = bool( row["kind"] == "host" and row["active"] and row["claim_handoff_id"] == row["handoff_id"] and row["lease_expires_at"] and current < _parse_utc(row["lease_expires_at"]) ) + if initial_only and prior_host_claim and live: + # Exact live recovery does not renew, mutate or increment a + # fence. The durable event retains the capability after a crash. + self.connection.execute("COMMIT") + return HandoffClaim( + run_id, row["handoff_id"], owner_id, row["fencing_token"], + row["lease_expires_at"], expected_version, "current", + ) if prior_claim is not None: if (not live or prior_claim.run_id != run_id or prior_claim.handoff_id != row["handoff_id"] @@ -4502,7 +4589,6 @@ def claim_handoff( else: if live: raise ConflictError("handoff already has a live claim") - prior_host_claim = row["kind"] == "host" and row["claim_handoff_id"] == row["handoff_id"] token = row["fencing_token"] + 1 action = "taken_over" if prior_host_claim else "acquired" self.connection.execute( @@ -4520,6 +4606,7 @@ def claim_handoff( "fencing_token": token, "handoff_id": row["handoff_id"], "owner_id": owner_id, + **({"terminal_finish": terminal_marker} if terminal_marker is not None else {}), }) self.connection.execute( "INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,?,?,?)", diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index fcd6819..344faa2 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -18,6 +18,7 @@ from devsquad.store import ( ConflictError, HandoffClaim, + SUPPORTED_SCHEMA_VERSION, Store, request_hash, ) @@ -197,7 +198,7 @@ def test_schema_four_fixture_migrates_to_host_handoffs(self): self.addCleanup(upgraded.close) self.assertEqual( upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], - 16, + SUPPORTED_SCHEMA_VERSION, ) tables = { row[0] diff --git a/test/core/test_terminal_ux.py b/test/core/test_terminal_ux.py index f910f82..d246f05 100644 --- a/test/core/test_terminal_ux.py +++ b/test/core/test_terminal_ux.py @@ -5,11 +5,13 @@ import io import json import os +import shlex from datetime import datetime, timedelta, timezone from pathlib import Path import subprocess import sys import tempfile +import time import unittest from unittest import mock @@ -20,7 +22,7 @@ from devsquad import detached from devsquad.contracts import ContractError from devsquad.service import Service -from devsquad.store import ConflictError, Store +from devsquad.store import ConflictError, Store, canonical_json, request_hash from devsquad.task_entry import build_managed_task @@ -61,7 +63,7 @@ def offline_start(task, *args): }} if task["workflow"] == "issue-delivery": fixtures["_internal_implementation_fixture"] = { - "writes": [{"path": "app.py", "content": "VALUE = 2\n"}], "delay_seconds": 0, + "iterations": [{"writes": [{"path": "app.py", "content": f"VALUE = {value}\n"}], "delay_seconds": 0} for value in (2, 3, 4)], } return original_start(task, *args, **fixtures) argv = [workflow] @@ -117,6 +119,11 @@ def test_normal_review_and_fix_finish_without_task_or_decision_json(self): for workflow in ("review", "fix"): with self.subTest(workflow=workflow): run = self.normal_run(workflow, workflow) + code, output, errors = self.invoke(["status", run]) + self.assertEqual((code, errors), (0, ""), output) + next_command = next(line.removeprefix("Next: ") for line in output.splitlines() if line.startswith("Next: ")) + self.assertEqual(shlex.split(next_command), ["squad", "finish", run, "--accept", "--reason", "your assessment of the saved evidence"]) + self.assertIn("Guidance: use --reject or --revise", output) code, output, errors = self.invoke(["finish", run, "--accept", "--reason", "Reviewed exact evidence."]) self.assertEqual((code, errors), (0, ""), output) self.assertIn("succeeded", output) @@ -139,6 +146,200 @@ def test_guided_finish_does_not_take_over_or_accept_failed_required_check(self): with self.assertRaisesRegex(ConflictError, "claim"): self.service.finish(run, "reject", "Another claim owns the packet.") + def interrupt_finish(self, run, *, service=None, reason="Reviewed exact evidence."): + service = service or self.service + captured = {} + def interrupt(run_id, claim, decision): + captured.update(claim=claim, decision=decision) + raise KeyboardInterrupt + with mock.patch.object(service, "handoff_complete", side_effect=interrupt): + with self.assertRaises(KeyboardInterrupt): + service.finish(run, "accept", reason) + self.assertEqual(service.status(run)["state"], "awaiting_host") + return captured + + def saved_artifacts(self, run): + store = self.service._store() + try: + return {item["id"]: (item, Path(item["path"]).read_bytes()) for item in store.artifacts_for_run(run)} + finally: + store.close() + + def assert_artifacts_unchanged(self, artifacts): + for artifact, content in artifacts.values(): + self.assertEqual(Path(artifact["path"]).read_bytes(), content) + + def test_interrupted_finish_recovers_its_exact_live_intent(self): + run = self.normal_run("interrupted-finish") + artifacts = self.saved_artifacts(run) + interrupted = self.interrupt_finish(run) + version = self.service.status(run)["version"] + store = self.service._store() + try: + payload = json.loads(store.connection.execute("SELECT payload FROM events WHERE run_id=? AND type='handoff.acquired' ORDER BY id DESC LIMIT 1", (run,)).fetchone()[0]) + self.assertEqual(payload["terminal_finish"]["decision"], interrupted["decision"]) + self.assertEqual(payload["terminal_finish"]["packet_sha256"], self.service.status(run)["handoff"]["packet_sha256"]) + finally: + store.close() + code, output, errors = self.invoke(["status", run]) + self.assertEqual((code, errors), (0, ""), output) + self.assertIn(f"squad finish {run} --accept --reason 'Reviewed exact evidence.'", output) + self.assertIn("retry the exact saved intent", output) + next_command = next(line.removeprefix("Next: ") for line in output.splitlines() if line.startswith("Next: ")) + self.assertEqual(shlex.split(next_command), ["squad", "finish", run, "--accept", "--reason", "Reviewed exact evidence."]) + code, output, _ = self.invoke(["status", run, "--json"]) + self.assertNotIn("handoff_view", json.loads(output)["data"]) + retried = self.interrupt_finish(run, service=Service(self.service.runtime)) + self.assertEqual(retried["claim"], interrupted["claim"]) + self.assertEqual(retried["decision"], interrupted["decision"]) + self.assertEqual(self.service.status(run)["version"], version) + self.assertEqual(Service(self.service.runtime).finish(run, "accept", "Reviewed exact evidence.")["state"], "succeeded") + self.assert_artifacts_unchanged(artifacts) + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "Reviewed exact evidence.") + + def test_expired_interrupted_finish_gets_only_its_own_fresh_fence(self): + run = self.normal_run("expired-finish") + artifacts = self.saved_artifacts(run) + interrupted = self.interrupt_finish(run) + later = datetime.fromisoformat(interrupted["claim"]["expires_at"]) + timedelta(seconds=1) + with mock.patch("devsquad.store._authoritative_now", return_value=later): + recovered = self.interrupt_finish(run, service=Service(self.service.runtime)) + self.assertEqual(recovered["claim"]["fencing_token"], interrupted["claim"]["fencing_token"] + 1) + self.assertEqual(recovered["decision"], interrupted["decision"]) + with self.assertRaises(ConflictError): + self.service.handoff_claim(run, self.service.status(run)["version"], "terminal-operator", interrupted["claim"]) + self.assertEqual(self.service.finish(run, "accept", "Reviewed exact evidence.")["state"], "succeeded") + self.assert_artifacts_unchanged(artifacts) + + def test_pending_terminal_finish_refuses_different_intent_and_corrupt_evidence(self): + run = self.normal_run("pending-finish") + interrupted = self.interrupt_finish(run) + before = self.service.status(run) + for disposition, reason in (("accept", "A changed reason."), ("reject", "Reviewed exact evidence.")): + for advance in (0, 700): + with mock.patch("devsquad.store._authoritative_now", return_value=datetime.now(timezone.utc) + timedelta(seconds=advance)): + with self.assertRaises(ConflictError): + self.service.finish(run, disposition, reason) + self.assertEqual(self.service.status(run)["version"], before["version"]) + artifacts = self.saved_artifacts(run) + artifact, content = artifacts[interrupted["decision"]["evidence_refs"][0]["artifact_id"]] + Path(artifact["path"]).write_bytes(b"corrupt pending evidence\n") + with self.assertRaisesRegex(ConflictError, "corrupt"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.status(run)["version"], before["version"]) + Path(artifact["path"]).write_bytes(content) + self.assertEqual(self.service.finish(run, "accept", interrupted["decision"]["reason"])["state"], "succeeded") + + def test_app_claim_named_terminal_operator_is_never_guided_authority(self): + run = self.normal_run("named-app-claim") + self.service.handoff_claim(run, self.service.status(run)["version"], "terminal-operator") + before = self.service.status(run) + for advance in (0, 700): + with mock.patch("devsquad.store._authoritative_now", return_value=datetime.now(timezone.utc) + timedelta(seconds=advance)): + with self.assertRaisesRegex(ConflictError, "claim"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.status(run)["version"], before["version"]) + + def test_later_app_claim_cannot_reuse_an_older_terminal_marker(self): + run = self.normal_run("superseded-terminal-claim") + interrupted = self.interrupt_finish(run) + later = datetime.fromisoformat(interrupted["claim"]["expires_at"]) + timedelta(seconds=1) + with mock.patch("devsquad.store._authoritative_now", return_value=later): + app_claim = self.service.handoff_claim(run, self.service.status(run)["version"], "terminal-operator") + self.assertGreater(app_claim["claim"]["fencing_token"], interrupted["claim"]["fencing_token"]) + with self.assertRaisesRegex(ConflictError, "claim"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + with mock.patch("devsquad.store._authoritative_now", return_value=later + timedelta(seconds=700)): + with self.assertRaisesRegex(ConflictError, "claim"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + + def test_terminal_intent_and_claim_roll_back_together_before_commit(self): + run = self.normal_run("atomic-finish-intent") + before = self.service.status(run) + def fail_marker(value): + if isinstance(value, dict) and "terminal_finish" in value: + raise KeyboardInterrupt + return canonical_json(value) + real_store = self.service._store + def interrupt_after_event(): + store = real_store() + connection = store.connection + class InterruptedConnection: + def __getattr__(self, name): + return getattr(connection, name) + def execute(self, sql, parameters=()): + result = connection.execute(sql, parameters) + if ("INSERT INTO events" in sql and len(parameters) >= 4 + and parameters[2] == "handoff.acquired" + and "terminal_finish" in json.loads(parameters[3])): + raise KeyboardInterrupt + return result + store.connection = InterruptedConnection() + return store + for point in ("before_event", "after_event"): + with self.subTest(point=point), contextlib.ExitStack() as patches: + if point == "before_event": + patches.enter_context(mock.patch("devsquad.store.canonical_json", side_effect=fail_marker)) + else: + patches.enter_context(mock.patch.object(self.service, "_store", side_effect=interrupt_after_event)) + with self.assertRaises(KeyboardInterrupt): + self.service.finish(run, "accept", "Reviewed exact evidence.") + after = self.service.status(run) + self.assertEqual(after["version"], before["version"]) + self.assertIsNone(after["handoff"]["claimed_by"]) + self.assertIsNone(self.service.handoff_view(run)["pending_finish"]) + self.assertEqual(self.service.finish(run, "accept", "Reviewed exact evidence.")["state"], "succeeded") + + def test_guided_finish_claim_is_fenced_by_a_concurrent_version_or_new_handoff(self): + for replacement in (False, True): + with self.subTest(replacement=replacement): + run = self.normal_run(f"finish-race-{replacement}", "fix" if replacement else "review") + self.addCleanup(self.service.cancel, run) + before = self.service.status(run) + real_claim = self.service.handoff_claim + competing = Service(self.service.runtime) + def race(run_id, version, owner, *args, **kwargs): + claim = competing.handoff_claim(run_id, version, "app-owner") + if replacement: + decision = copy.deepcopy(kwargs["_terminal_decision"]) + decision.update(disposition="revise", reason="Review the bounded revision.") + decision["submission_hash"] = request_hash({key: value for key, value in decision.items() if key != "submission_hash"}) + competing.handoff_complete(run_id, claim["claim"], decision) + real_status = competing.status + deadline = time.monotonic() + 30 + def bounded_status(selected_run): + status = real_status(selected_run) + self.assertLess(time.monotonic(), deadline, status) + return status + with mock.patch.object(competing, "status", side_effect=bounded_status): + response, code = cli._wait_for_run(competing, {"run_id": run_id}, resume_candidate_review=True) + self.assertEqual(code, 2, response) + self.assertNotEqual(competing.status(run_id)["handoff"]["handoff_id"], before["handoff"]["handoff_id"]) + return real_claim(run_id, version, owner, *args, **kwargs) + with mock.patch.object(self.service, "handoff_claim", side_effect=race): + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.status(run)["state"], "awaiting_host") + + def test_saved_finish_submission_still_resumes_the_existing_continuation(self): + run = self.normal_run("saved-finish-submission") + with mock.patch.object(self.service, "_continue_branch_review_submission", side_effect=KeyboardInterrupt): + with self.assertRaises(KeyboardInterrupt): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.status(run)["next_action"], "handoff_submission_saved") + self.assertEqual(Service(self.service.runtime).resume(run)["state"], "succeeded") + + def test_cancelled_pending_terminal_finish_never_recovers_or_replays(self): + run = self.normal_run("cancelled-finish") + self.interrupt_finish(run) + self.assertEqual(self.service.cancel(run)["state"], "cancelled") + for advance in (0, 700): + with mock.patch("devsquad.store._authoritative_now", return_value=datetime.now(timezone.utc) + timedelta(seconds=advance)): + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "Reviewed exact evidence.") + self.assertEqual(self.service.result(run)["state"], "cancelled") + def test_reject_and_revision_exhaustion_use_the_existing_terminal_gates(self): rejected = self.normal_run("reject") decision = self.service.finish(rejected, "reject", "Review remains incomplete.") From 7130ee7689f8cc14efacb606bcbc4c05f373298c Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 16:24:12 -0700 Subject: [PATCH 183/197] WIP checkpoint: R6: finish audited expiry recovery and verified command round trips (2026-10-02 16:24) --- docs/RUNTIME-GUIDE.md | 8 +- docs/plans/engineering-team/RESUME.md | 30 ++- ...terminal-readiness-partial-2026-10-02.json | 21 ++- plugin/core/src/devsquad/cli.py | 4 +- plugin/core/src/devsquad/store.py | 95 ++++++++-- test/core/test_handoff_store.py | 177 ++++++++++++++++++ test/core/test_terminal_ux.py | 73 +++++++- 7 files changed, 375 insertions(+), 33 deletions(-) diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index 13913fc..5937dd6 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -108,7 +108,13 @@ claim, finish saves the exact decision atomically with that claim. Status shows the exact retry command. Only that same packet, disposition and reason can recover the live claim or its expired fence; an owner name alone never permits recovery. If the decision was already submitted, use `squad resume RUN_ID` to -finish the saved continuation. The normal lead is the terminal host; +finish the saved continuation. A claim that expires just before submission is +still rejected and audited. An exact guided retry under its own fresh valid +fence can recover that expiry: the unique submission row is an operational +projection, while an append-only recovery event preserves the complete original +rejected row and its digest. Final event exports retain both rejection and +recovery history. App claims and other rejection reasons cannot use this path. +The normal lead is the terminal host; an explicitly configured headless run follows its existing lead through a temporary handoff when observed with `--wait`. Omitting `--wait` returns the run ID immediately. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 9dd6254..be7e598 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -109,11 +109,18 @@ assertion. Root's integrated 61 terminal/CLI/store/handoff tests pass in repair tests/26.811s but found two edge cases: an `expired_claim` rejection poisons identical retry even after its own fresh fence, and a reason beginning with `--` makes the emitted separate `--reason` argument invalid. The terminal -agent owns the narrow follow-up: preserve the full rejected row canonically in -append-only audit history before a gated operational-row recovery; keep all -ordinary app replay rules unchanged. Attach the quoted reason with `--reason=`. -Integrate/review/test this follow-up before one frozen full suite and exact -native R6 audit. No full/native gate has started. Production remains accepted +agent repaired these at `31a8a68b`: complete rejected-row/digest immutable audit +and exact fresh-authority recovery, consecutive event versions, unchanged app +replay and actual `--reason=` CLI round-trip. Its 156-test affected gate passes +in 350.005s, no skips/warnings; independent six-test/9.120s review is clean and +exact reviewed blobs/diff match the checkpoint. Final `e7e9042c` additionally +requires the canonical original rejection event to match its row/run/version/ +timestamp/ID/hash/reason; 19 focused tests/19.207s pass, 13 negative subcases. +Both follow-ups are integrated; root's final 19 focused tests pass in 7.183s. +Bash/reference/diff gates pass. Final added-hunk independent review is clean; +one test/13 negative subcases in 0.208s, reviewed/committed blobs equivalent. +Next freeze one full suite and exact native R6 audit, then safe install only +after acceptance. No full/native gate has started. Production remains accepted R5/schema16. R7 Council remains in isolated `r7-council`; its controlled stage flow is partial. Its source is checkpointed at `42979ba4` with 21 Council tests and @@ -126,8 +133,17 @@ the immutable installed R5/schema16 client can acquire a Council headless handoff that the new client rejects, stranding its completed lead. Both stores were authoritatively schema16; temporary runs were cancelled and owned worker PIDs confirmed absent. The Council agent owns a minimal schema17 compatibility -epoch and old-client/active-upgrade tests. No production edit or new native -Council generation has been made. Preserve all attached worktrees. +epoch and old-client/active-upgrade tests. Residual checkpoint `72ae4130` now +has the minimal no-table schema17 epoch, owned gated-launch cancellation and +actual matched/held-out fixture workflow comparison: inconclusive, all native +quality/escaped-defect/rework/quota/host-usage observations remain unknown, +automatic use off. Its 25 Council tests/90.724s, 35 shared migration tests/ +54.469s, 12 handoff-store tests/4.528s and 10 supervisor/crash tests/1.819s pass. +Native HTTPS attestation fails closed before generation. The Council agent +owns actual temporary immutable16-to17 install-upgrade proof; readiness owns +immutable-old16 claim denial and one bounded Unix mDNS resolver-socket diagnostic. +Root owns shared-hunk/formatter reconciliation after R6 acceptance. No production +edit or new native Council generation has occurred. Preserve attached worktrees. Desktop control worked for scoped inspection. Claude's local Code tab selected only this DevSquad project on `codex/engineering-team`, with an empty prompt; no proof diff --git a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json index 250a74d..2011419 100644 --- a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json @@ -2,7 +2,7 @@ "schema_version": 1, "work_package": "R6", "recorded_on": "2026-10-02", - "status": "recovery_repair_integrated_followup_required", + "status": "recovery_repairs_integrated_full_native_pending", "base_revision": "8f2c1c6a7d9edfe7cc3d78ecbc59ce105be636b2", "integrated_checkpoints": [ "dd709804e70e0791614319a9ae5c2597768d7318", @@ -11,14 +11,17 @@ "d0eec55c9dbede205e3c8831b7c224d3d77114d3", "4a6bc15d81017aa996be9a21e6c81bc02ebd17a6", "03124c79ba53cb1b12296bb554024605e4cadbc7", - "22a4ed8ac85b95861f4ba9725ca729388b7a9f49" + "22a4ed8ac85b95861f4ba9725ca729388b7a9f49", + "31a8a68b59db3d0af136c5c18cb67fe5088c02b0", + "e7e9042cbf4d1416209cf39e0269f7fa85479f4c" ], "contract": "SOL-REVIEW-FOLLOWUP R6/G3/G4; unchanged low-level MCP envelopes, fenced host disposition, Bash 3.2 and optional jq", "implemented": [ "Readable normal terminal commands; --json retains the versioned automation envelope", "Guided finish binds the exact current packet/artifact hashes and existing independent-review/check/claim fences; rejects replay and app-owned prior host claims", "Finish intent and claim commit atomically; only an identical canonical decision whose latest acquisition event matches the current packet/owner/fence/expiry recovers its live or expired guided claim. App claims with the same owner name and later renewals cannot reuse old terminal authority", - "Saved submissions continue through resume; human status prints a copyable, shell-quoted exact intent retry command with separate guidance", + "Saved submissions continue through resume; human status prints a copyable, shell-quoted --reason= exact intent retry command with separate guidance, including negative-leading reasons", + "A prior authoritative expired_claim rejection can recover only with exact latest guided intent and a fresh live matching claim; the complete rejected row and digest are atomically preserved in append-only history before the operational projection becomes recorded. Original rejection and recovery/submission consecutive event versions remain visible; no schema or decision-ID rewrite", "Omitted IDs require exactly one canonical current-project run; zero and multiple choices are explicit; unrelated runs never selected", "Doctor distinguishes installed, registered, exact supported version, subscription authentication, workflow readiness and unknown operation proof", "Bounded non-generating Claude auth and Codex account/read with no API keys or alternate credential-home overrides", @@ -33,13 +36,16 @@ {"revision": "4a6bc15d", "tests": 41, "seconds": 11.356, "optional_sdk_skips": 2, "bash_assertions": 227}, {"revision": "03124c79", "tests": 106, "seconds": 95.408, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass"}, {"revision": "22a4ed8a", "tests": 150, "seconds": 180.750, "failures": 0, "errors": 0, "scope": "Affected repair gate before final command-copyability edit"}, - {"revision": "22a4ed8a", "tests": 45, "seconds": 48.316, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Final CLI/terminal repair checkpoint"} + {"revision": "22a4ed8a", "tests": 45, "seconds": 48.316, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Final CLI/terminal repair checkpoint"}, + {"revision": "31a8a68b", "tests": 156, "seconds": 350.005, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Expiry rejection and command round-trip repair, before final original-rejection-event crosscheck"}, + {"revision": "e7e9042c", "tests": 19, "seconds": 19.207, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Final original-rejection-event crosscheck; 13 negative audit-corruption subcases"} ], "root_affected_gates": [ {"command": "python3 -m unittest test_diagnostics test_terminal_ux test_installed_usability test_cli test_task_entry test_mcp test_handoff_store", "tests": 114, "seconds": 65.066, "optional_sdk_skips": 2, "failures": 0, "errors": 0, "scope": "Integrated first five checkpoints, before final catalog cleanup reuse"}, {"command": "python3 -m unittest test_delivery_workflow test_delivery_identity_runtime test_handoff_service", "tests": 32, "seconds": 66.877, "failures": 0, "errors": 0}, {"command": "python3 -m unittest test_task_entry test_installed_usability", "tests": 27, "seconds": 25.389, "skips": 0, "failures": 0, "errors": 0, "scope": "All six integrated checkpoints including final catalog cleanup reuse"}, - {"command": "python3 -m unittest test_terminal_ux test_cli test_handoff_store test_handoff_service", "tests": 61, "seconds": 55.963, "failures": 0, "errors": 0, "scope": "Final integrated interrupted-finish repair and current-schema assertion; ResourceWarning strict"} + {"command": "python3 -m unittest test_terminal_ux test_cli test_handoff_store test_handoff_service", "tests": 61, "seconds": 55.963, "failures": 0, "errors": 0, "scope": "Integrated initial interrupted-finish repair and current-schema assertion; ResourceWarning strict"}, + {"command": "python3 -m unittest test_handoff_store test_terminal_ux.TerminalUxTest.test_finish_retries_authoritative_expiry_rejection_with_exact_fresh_claim test_terminal_ux.TerminalUxTest.test_pending_finish_negative_leading_reason_round_trips_as_actual_cli", "tests": 19, "seconds": 7.183, "failures": 0, "errors": 0, "scope": "All recovery follow-ups including final exact original rejection audit; ResourceWarning strict"} ], "failure_history": [ "Readiness red baseline: nine tests, three failures and 17 subtest errors before implementation", @@ -53,8 +59,10 @@ "Repair red reproduction: three selected tests had one failure and two errors before durable intent implementation", "First expanded repair-agent gate failed a stale isolated bf3 current-schema expectation of 15; root already expected 16 after dfe9976. Changed only the current-schema assertion to SUPPORTED_SCHEMA_VERSION and retained the historical schema-four fixture checks; not a current root runtime failure", "Independent 22a4ed8a follow-up: ten existing repair tests pass in 26.811s, but a lease expiring after claim acquisition and before submission creates a durable expired_claim rejection. A fresh own retry advances fence 2 to 3/version 16 to 17 yet the stable decision replays that old rejection indefinitely. Follow-up must preserve complete immutable rejection history while recovering only exact fresh guided authority", - "Independent 22a4ed8a command-copyability edge: saved reason --deferred is printed as --reason --deferred and argparse rejects it. Attach the shell-quoted reason to --reason= and test actual command round-trip" + "Independent 22a4ed8a command-copyability edge: saved reason --deferred is printed as --reason --deferred and argparse rejects it. Fixed by 31a8a68b attached shell-quoted --reason= and actual command round-trip", + "Follow-up red reproduction: two tests/3.845s, one failure and one error. Initial audit implementation reused a run_version for two events and failed uniqueness; consecutive atomic recovery/submitted versions repaired it before green checkpoint gates" ], + "independent_followup": {"revision": "31a8a68b", "result": "clean_bounded_scope", "tests": 6, "seconds": 9.120, "source_blobs_before_after": "unchanged", "full_five_file_diff_sha256": "b4c742fa400932d216d698e483af40abaa733c741325aebd3eaaf56776accf40", "final_original_event_hunk_review": {"revision": "e7e9042c", "result": "clean_bounded_scope", "tests": 1, "negative_subcases": 13, "seconds": 0.208, "source_before_after_committed": "equivalent", "two_file_diff_sha256": "39ba13f6e5b0cf06da5c55d619d7bd0b2a3684780129d16f8493640736b8cc65"}}, "fresh_install_scope": { "providers": "offline native CLI protocol fixtures only", "installer": "actual source installer with no-index and temporary HOME/install/runtime/bin", @@ -75,7 +83,6 @@ "unrelated_servers_or_settings": "not changed" }, "remaining_gates": [ - "Expiry-rejection poisoning and leading-hyphen reason follow-up, final integrated regressions and independent repair review", "Frozen full gate", "Independent exact R6 review and accepted immutable checkpoint", "Safe installed update, non-generating live readiness, idempotence/drift and actual installed SDK tests", diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index 2790077..bd192bd 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -660,11 +660,11 @@ def _next_command(data: dict[str, Any]) -> str: if action == "claim_handoff": pending = (data.get("handoff_view") or {}).get("pending_finish") if pending: - return f"squad finish {run_id} --{pending['disposition']} --reason {shlex.quote(pending['reason'])}" + return f"squad finish {run_id} --{pending['disposition']} --reason={shlex.quote(pending['reason'])}" owner = (data.get("handoff") or {}).get("claimed_by") if owner: return f"complete or renew the saved claim in {owner}; this handoff already has an owner" - return f'squad finish {run_id} --accept --reason "your assessment of the saved evidence"' + return f'squad finish {run_id} --accept --reason="your assessment of the saved evidence"' if action in {"continue_headless_lead", "resume_candidate_review", "handoff_submission_saved"}: return f"squad resume {run_id}" if action == "recovery_file_required": diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 70e417e..8fcec09 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -4728,9 +4728,54 @@ def record_handoff_submission( if reused_id and rejection is None: rejection = "submission_id_reused" persisted_claim = self.connection.execute( - "SELECT kind,owner_id,fencing_token,active,handoff_id,lease_expires_at " - "FROM claims WHERE run_id=?", (run_id,), + "SELECT c.kind,c.owner_id,c.fencing_token,c.active,c.lease_expires_at," + "c.handoff_id AS claim_handoff_id,h.id AS handoff_id,h.packet_sha256 " + "FROM claims c JOIN handoffs h ON h.run_id=c.run_id " + "WHERE c.run_id=? AND h.id=?", (run_id, claim.handoff_id), ).fetchone() + recovered_rejection = None + if (prior is not None and prior["outcome"] == "rejected" + and isinstance(prior["id"], str) and prior["id"] + and prior["rejection_code"] == "expired_claim" + and claim.owner_id == "terminal-operator" + and prior["owner_id"] == claim.owner_id + and type(prior["fencing_token"]) is int + and prior["fencing_token"] < claim.fencing_token + and prior["decision_json"] == decision_json + and prior["evidence_refs_json"] == evidence_json + and prior["disposition"] == disposition + and type(prior["recorded_run_version"]) is int + and prior["recorded_run_version"] < run["version"] + and reused_id is None + and persisted_claim is not None + and claim.expires_at == persisted_claim["lease_expires_at"]): + saved = self._terminal_finish_for_claim(run_id, persisted_claim) + if saved is not None and canonical_json(saved) == decision_json: + rejected_event = self.connection.execute( + "SELECT run_id,run_version,type,payload,created_at FROM events " + "WHERE run_id=? AND run_version=?", + (run_id, prior["recorded_run_version"]), + ).fetchone() + expected_payload = canonical_json({ + "handoff_id": prior["handoff_id"], + "reason": "expired_claim", + "submission_hash": prior["submission_hash"], + "submission_id": prior["submission_id"], + }) + if (rejected_event is None + or rejected_event["run_id"] != run_id + or rejected_event["run_version"] != prior["recorded_run_version"] + or rejected_event["type"] != "handoff.completion_rejected" + or rejected_event["created_at"] != prior["created_at"] + or rejected_event["payload"] != expected_payload): + raise ConflictError("expired terminal finish rejection audit is invalid") + # The row is an operational projection with a unique + # decision identity. Preserve its complete rejected state + # in the append-only log before changing that projection. + if _parse_utc(prior["created_at"]) > current: + raise ConflictError("expired terminal finish rejection timestamp is invalid") + recovered_rejection = dict(prior) + rejection = None if rejection is None: if run["state"] in TERMINAL_STATES: rejection = "terminal_run" @@ -4739,7 +4784,7 @@ def record_handoff_submission( rejection = "handoff_not_open" elif (not persisted_claim or persisted_claim["kind"] != "host" or not persisted_claim["active"] - or persisted_claim["handoff_id"] != claim.handoff_id + or persisted_claim["claim_handoff_id"] != claim.handoff_id or persisted_claim["owner_id"] != claim.owner_id or persisted_claim["fencing_token"] != claim.fencing_token): rejection = "stale_claim" @@ -4748,16 +4793,40 @@ def record_handoff_submission( rejection = "expired_claim" if rejection is None: version = run["version"] + 1 - submission_row_id = str(uuid.uuid4()) - self.connection.execute( - "INSERT INTO handoff_submissions(id,handoff_id,submission_id,submission_hash," - "owner_id,fencing_token,disposition,decision_json,evidence_refs_json,outcome," - "rejection_code,recorded_run_version,created_at) " - "VALUES(?,?,?,?,?,?,?,?,?,'recorded',NULL,?,?)", - (submission_row_id, claim.handoff_id, submission_id, submission_hash, - claim.owner_id, claim.fencing_token, disposition, decision_json, - evidence_json, version, timestamp), - ) + if recovered_rejection is not None: + payload = canonical_json({ + "disposition": disposition, + "handoff_id": claim.handoff_id, + "fencing_token": claim.fencing_token, + "submission_hash": submission_hash, + "submission_id": submission_id, + "rejected_submission": recovered_rejection, + "rejected_submission_sha256": request_hash(recovered_rejection), + }) + self.connection.execute( + "INSERT INTO events(run_id,run_version,type,payload,created_at) " + "VALUES(?,?,'handoff.completion_recovered',?,?)", + (run_id, version, payload, timestamp), + ) + version += 1 + updated = self.connection.execute( + "UPDATE handoff_submissions SET owner_id=?,fencing_token=?,outcome='recorded'," + "rejection_code=NULL,recorded_run_version=?,created_at=? " + "WHERE id=? AND outcome='rejected' AND rejection_code='expired_claim'", + (claim.owner_id, claim.fencing_token, version, timestamp, prior["id"]), + ) + if updated.rowcount != 1: + raise ConflictError("expired terminal finish projection changed") + else: + self.connection.execute( + "INSERT INTO handoff_submissions(id,handoff_id,submission_id,submission_hash," + "owner_id,fencing_token,disposition,decision_json,evidence_refs_json,outcome," + "rejection_code,recorded_run_version,created_at) " + "VALUES(?,?,?,?,?,?,?,?,?,'recorded',NULL,?,?)", + (str(uuid.uuid4()), claim.handoff_id, submission_id, submission_hash, + claim.owner_id, claim.fencing_token, disposition, decision_json, + evidence_json, version, timestamp), + ) self.connection.execute( "UPDATE handoffs SET status='submitted',submitted_run_version=?,closed_at=? " "WHERE id=?", (version, timestamp, claim.handoff_id), diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 344faa2..31a5512 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -20,6 +20,7 @@ HandoffClaim, SUPPORTED_SCHEMA_VERSION, Store, + canonical_json, request_hash, ) from devsquad.contracts import ContractError @@ -488,6 +489,182 @@ def test_expired_submission_is_audited_and_cannot_displace_takeover(self): self.assertEqual(current_claim.owner_id, takeover.owner_id) self.assertEqual(current_claim.fencing_token, takeover.fencing_token) + def guided_expiry_recovery(self, key, *, guided=True): + run_id, _, snapshot = self.waiting_run(key) + started = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) + decision = self.decision(submission_id=f"terminal-{snapshot.handoff_id}") + options = {"initial_only": True, "terminal_decision": decision} if guided else {} + old = self.store.claim_handoff(run_id, snapshot.run_version, "terminal-operator", now=started, **options) + later = datetime.fromisoformat(old.expires_at) + timedelta(seconds=1) + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.store.record_handoff_submission(run_id, old, decision, now=later) + rejected = dict(self.store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (old.handoff_id,)).fetchone()) + current = self.store.claim_handoff(run_id, self.store.run(run_id)["version"], "terminal-operator", now=later, **options) + return run_id, old, current, decision, later, rejected + + def handoff_rows(self, run_id): + return {table: [dict(row) for row in self.store.connection.execute(query, (run_id,))] for table, query in { + "runs": "SELECT * FROM runs WHERE id=?", + "claims": "SELECT * FROM claims WHERE run_id=?", + "handoffs": "SELECT * FROM handoffs WHERE run_id=?", + "submissions": "SELECT s.* FROM handoff_submissions s JOIN handoffs h ON h.id=s.handoff_id WHERE h.run_id=?", + "events": "SELECT * FROM events WHERE run_id=? ORDER BY id", + }.items()} + + def test_guided_expiry_audit_and_projection_commit_or_roll_back_together(self): + for interruption in ("before-audit", "after-audit", "after-projection"): + with self.subTest(interruption=interruption): + run_id, old, current, decision, later, rejected = self.guided_expiry_recovery(interruption) + before = self.handoff_rows(run_id) + connection = self.store.connection + class InterruptedConnection: + def execute(self, statement, *args, **kwargs): + audit = "'handoff.completion_recovered'" in statement + projection = statement.startswith("UPDATE handoff_submissions SET") + if audit and interruption == "before-audit": + raise KeyboardInterrupt + result = connection.execute(statement, *args, **kwargs) + if (audit and interruption == "after-audit") or (projection and interruption == "after-projection"): + raise KeyboardInterrupt + return result + def __getattr__(self, name): + return getattr(connection, name) + self.store.connection = InterruptedConnection() + try: + with self.assertRaises(KeyboardInterrupt): + self.store.record_handoff_submission(run_id, current, decision, now=later) + finally: + self.store.close() + self.store = Store(self.database, self.artifacts) + self.assertEqual(self.handoff_rows(run_id), before) + recorded = self.store.record_handoff_submission(run_id, current, decision, now=later) + self.assertFalse(recorded.replayed) + after = self.handoff_rows(run_id) + self.assertEqual(after["events"][:-2], before["events"]) + audit_event, submitted = after["events"][-2:] + self.assertEqual((audit_event["type"], submitted["type"]), ("handoff.completion_recovered", "handoff.submitted")) + payload = json.loads(audit_event["payload"]) + self.assertEqual(audit_event["payload"], canonical_json(payload)) + self.assertEqual(payload["rejected_submission"], rejected) + self.assertEqual(payload["rejected_submission_sha256"], request_hash(rejected)) + self.assertEqual(submitted["run_version"], audit_event["run_version"] + 1) + self.assertEqual(recorded.recorded_run_version, submitted["run_version"]) + self.assertEqual(after["runs"][0]["version"], submitted["run_version"]) + self.assertEqual(after["handoffs"][0]["submitted_run_version"], submitted["run_version"]) + self.assertEqual(after["submissions"][0]["recorded_run_version"], submitted["run_version"]) + replay = self.store.record_handoff_submission(run_id, old, decision, now=later + timedelta(hours=1)) + self.assertTrue(replay.replayed) + self.assertEqual(replay.recorded_run_version, recorded.recorded_run_version) + self.assertEqual(self.handoff_rows(run_id), after) + + def test_guided_expiry_recovery_refuses_malformed_prior_row_or_marker(self): + corruptions = { + "decision": ("decision_json", "{}"), + "evidence": ("evidence_refs_json", "[{}]"), + "owner": ("owner_id", "other-app"), + "disposition": ("disposition", "reject"), + "rejection": ("rejection_code", "stale_claim"), + "created": ("created_at", "not-a-time"), + "future-created": ("created_at", "2099-01-01T00:00:00+00:00"), + "same-fence": ("fencing_token", None), + "same-version": ("recorded_run_version", None), + "marker-structure": None, + "marker-packet": None, + } + for name, corruption in corruptions.items(): + with self.subTest(corruption=name): + run_id, _, current, decision, later, rejected = self.guided_expiry_recovery(name) + if corruption is None: + event = self.store.connection.execute("SELECT id,payload FROM events WHERE run_id=? AND type='handoff.taken_over' ORDER BY id DESC LIMIT 1", (run_id,)).fetchone() + payload = json.loads(event["payload"]) + if name == "marker-structure": + payload["terminal_finish"]["unexpected"] = True + else: + payload["terminal_finish"]["packet_sha256"] = "0" * 64 + self.store.connection.execute("UPDATE events SET payload=? WHERE id=?", (canonical_json(payload), event["id"])) + else: + field, value = corruption + if name == "same-fence": + value = current.fencing_token + elif name == "same-version": + value = current.run_version + self.store.connection.execute(f"UPDATE handoff_submissions SET {field}=? WHERE id=?", (value, rejected["id"])) + before = self.handoff_rows(run_id) + with self.assertRaises(ConflictError): + self.store.record_handoff_submission(run_id, current, decision, now=later) + self.assertEqual(self.handoff_rows(run_id), before) + + def test_guided_expiry_recovery_requires_current_marker_not_same_name_app(self): + for operation in ("ordinary-app", "renew", "takeover", "changed-intent"): + with self.subTest(operation=operation): + run_id, _, current, decision, later, _ = self.guided_expiry_recovery(operation, guided=operation != "ordinary-app") + if operation == "renew": + current = self.store.claim_handoff(run_id, current.run_version, "terminal-operator", current, now=later + timedelta(seconds=1)) + later += timedelta(seconds=1) + elif operation == "takeover": + later = datetime.fromisoformat(current.expires_at) + timedelta(seconds=1) + current = self.store.claim_handoff(run_id, current.run_version, "terminal-operator", now=later) + elif operation == "changed-intent": + decision = self.decision(submission_id=decision["submission_id"], reason="different exact intent") + before = self.handoff_rows(run_id) + with self.assertRaises(ConflictError): + self.store.record_handoff_submission(run_id, current, decision, now=later) + after = self.handoff_rows(run_id) + self.assertFalse(any(row["type"] == "handoff.completion_recovered" for row in after["events"])) + self.assertEqual(after["submissions"][0], before["submissions"][0]) + if operation != "changed-intent": + self.assertEqual(after, before) + self.assertIsNone(self.store.terminal_finish_decision(run_id, current.handoff_id)) + + def test_guided_expiry_recovery_requires_exact_original_rejection_event(self): + for corruption in ("missing", "type", "created", "row-created", "row-version", "run", "version", "noncanonical", "extra", "handoff", "submission-id", "hash", "reason"): + with self.subTest(corruption=corruption): + run_id, _, current, decision, later, rejected = self.guided_expiry_recovery(f"audit-{corruption}") + event = self.store.connection.execute("SELECT * FROM events WHERE run_id=? AND run_version=?", (run_id, rejected["recorded_run_version"])).fetchone() + if corruption == "missing": + self.store.connection.execute("DELETE FROM events WHERE id=?", (event["id"],)) + elif corruption == "type": + self.store.connection.execute("UPDATE events SET type='handoff.submitted' WHERE id=?", (event["id"],)) + elif corruption == "created": + self.store.connection.execute("UPDATE events SET created_at=? WHERE id=?", ((later + timedelta(seconds=1)).isoformat(), event["id"])) + elif corruption == "row-created": + self.store.connection.execute("UPDATE handoff_submissions SET created_at=? WHERE id=?", ((later - timedelta(seconds=1)).isoformat(), rejected["id"])) + elif corruption == "row-version": + self.store.connection.execute("UPDATE handoff_submissions SET recorded_run_version=? WHERE id=?", (rejected["recorded_run_version"] - 1, rejected["id"])) + elif corruption == "run": + other, _, _ = self.waiting_run("audit-other-run") + self.store.connection.execute("UPDATE events SET run_id=? WHERE id=?", (other, event["id"])) + elif corruption == "version": + self.store.connection.execute("UPDATE events SET run_version=? WHERE id=?", (current.run_version + 100, event["id"])) + elif corruption == "noncanonical": + self.store.connection.execute("UPDATE events SET payload=? WHERE id=?", (json.dumps(json.loads(event["payload"])), event["id"])) + else: + payload = json.loads(event["payload"]) + if corruption == "extra": + payload["extra"] = True + else: + field = {"handoff": "handoff_id", "submission-id": "submission_id", "hash": "submission_hash", "reason": "reason"}[corruption] + payload[field] = "different" + self.store.connection.execute("UPDATE events SET payload=? WHERE id=?", (canonical_json(payload), event["id"])) + before = self.handoff_rows(run_id) + with self.assertRaisesRegex(ConflictError, "rejection audit"): + self.store.record_handoff_submission(run_id, current, decision, now=later) + self.assertEqual(self.handoff_rows(run_id), before) + + def test_repeated_guided_expiry_retains_one_rejection_and_one_recovery(self): + run_id, _, current, decision, later, rejected = self.guided_expiry_recovery("repeated-expiry") + again = datetime.fromisoformat(current.expires_at) + timedelta(seconds=1) + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.store.record_handoff_submission(run_id, current, decision, now=again) + self.assertEqual(dict(self.store.connection.execute("SELECT * FROM handoff_submissions WHERE id=?", (rejected["id"],)).fetchone()), rejected) + latest = self.store.claim_handoff(run_id, self.store.run(run_id)["version"], "terminal-operator", now=again, initial_only=True, terminal_decision=decision) + self.assertGreater(latest.fencing_token, current.fencing_token) + self.store.record_handoff_submission(run_id, latest, decision, now=again) + rows = self.handoff_rows(run_id) + self.assertEqual(sum(row["type"] == "handoff.completion_rejected" for row in rows["events"]), 1) + self.assertEqual(sum(row["type"] == "handoff.completion_recovered" for row in rows["events"]), 1) + self.assertEqual(len(rows["submissions"]), 1) + def test_submission_replays_and_terminal_late_rejection_preserves_run_and_events(self): run_id, _, snapshot = self.waiting_run() claimed_at = datetime(2026, 9, 15, 5, 1, tzinfo=timezone.utc) diff --git a/test/core/test_terminal_ux.py b/test/core/test_terminal_ux.py index d246f05..3fb4479 100644 --- a/test/core/test_terminal_ux.py +++ b/test/core/test_terminal_ux.py @@ -122,7 +122,7 @@ def test_normal_review_and_fix_finish_without_task_or_decision_json(self): code, output, errors = self.invoke(["status", run]) self.assertEqual((code, errors), (0, ""), output) next_command = next(line.removeprefix("Next: ") for line in output.splitlines() if line.startswith("Next: ")) - self.assertEqual(shlex.split(next_command), ["squad", "finish", run, "--accept", "--reason", "your assessment of the saved evidence"]) + self.assertEqual(shlex.split(next_command), ["squad", "finish", run, "--accept", "--reason=your assessment of the saved evidence"]) self.assertIn("Guidance: use --reject or --revise", output) code, output, errors = self.invoke(["finish", run, "--accept", "--reason", "Reviewed exact evidence."]) self.assertEqual((code, errors), (0, ""), output) @@ -183,10 +183,10 @@ def test_interrupted_finish_recovers_its_exact_live_intent(self): store.close() code, output, errors = self.invoke(["status", run]) self.assertEqual((code, errors), (0, ""), output) - self.assertIn(f"squad finish {run} --accept --reason 'Reviewed exact evidence.'", output) + self.assertIn(f"squad finish {run} --accept --reason='Reviewed exact evidence.'", output) self.assertIn("retry the exact saved intent", output) next_command = next(line.removeprefix("Next: ") for line in output.splitlines() if line.startswith("Next: ")) - self.assertEqual(shlex.split(next_command), ["squad", "finish", run, "--accept", "--reason", "Reviewed exact evidence."]) + self.assertEqual(shlex.split(next_command), ["squad", "finish", run, "--accept", "--reason=Reviewed exact evidence."]) code, output, _ = self.invoke(["status", run, "--json"]) self.assertNotIn("handoff_view", json.loads(output)["data"]) retried = self.interrupt_finish(run, service=Service(self.service.runtime)) @@ -212,6 +212,73 @@ def test_expired_interrupted_finish_gets_only_its_own_fresh_fence(self): self.assertEqual(self.service.finish(run, "accept", "Reviewed exact evidence.")["state"], "succeeded") self.assert_artifacts_unchanged(artifacts) + def test_finish_retries_authoritative_expiry_rejection_with_exact_fresh_claim(self): + run = self.normal_run("finish-expired-at-completion") + artifacts = self.saved_artifacts(run) + original_complete = self.service.handoff_complete + captured = {} + def expire_before_submission(run_id, claim, decision): + captured.update(claim=claim, decision=decision) + later = datetime.fromisoformat(claim["expires_at"]) + timedelta(seconds=1) + captured["later"] = later + with mock.patch("devsquad.store._authoritative_now", return_value=later): + return original_complete(run_id, claim, decision) + with mock.patch.object(self.service, "handoff_complete", side_effect=expire_before_submission): + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.service.finish(run, "accept", "Reviewed exact evidence.") + store = self.service._store() + try: + rejected = dict(store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (captured["claim"]["handoff_id"],)).fetchone()) + events_before = [dict(row) for row in store.connection.execute("SELECT * FROM events WHERE run_id=? ORDER BY id", (run,))] + finally: + store.close() + self.assertEqual((rejected["outcome"], rejected["rejection_code"]), ("rejected", "expired_claim")) + self.assertEqual(self.service.status(run)["state"], "awaiting_host") + with mock.patch("devsquad.store._authoritative_now", return_value=captured["later"]): + recovered = self.interrupt_finish(run, service=Service(self.service.runtime)) + self.assertEqual(recovered["decision"], captured["decision"]) + self.assertEqual(recovered["claim"]["fencing_token"], captured["claim"]["fencing_token"] + 1) + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.service.handoff_complete(run, captured["claim"], captured["decision"]) + self.assertEqual(Service(self.service.runtime).finish(run, "accept", "Reviewed exact evidence.")["state"], "succeeded") + store = self.service._store() + try: + events_after = [dict(row) for row in store.connection.execute("SELECT * FROM events WHERE run_id=? ORDER BY id", (run,))] + audit = json.loads(store.connection.execute("SELECT payload FROM events WHERE run_id=? AND type='handoff.completion_recovered'", (run,)).fetchone()[0]) + recorded = dict(store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (captured["claim"]["handoff_id"],)).fetchone()) + finally: + store.close() + self.assertEqual(events_after[:len(events_before)], events_before) + self.assertEqual(audit["rejected_submission"], rejected) + self.assertEqual(audit["rejected_submission_sha256"], request_hash(rejected)) + self.assertEqual(audit["fencing_token"], recovered["claim"]["fencing_token"]) + self.assertEqual((recorded["outcome"], recorded["rejection_code"]), ("recorded", None)) + self.assertEqual(recorded["decision_json"], rejected["decision_json"]) + self.assertEqual(recorded["evidence_refs_json"], rejected["evidence_refs_json"]) + result = self.service.result(run) + exported_path = next(Path(item["path"]) for item in result["artifacts"] if item["name"] == "events.jsonl") + exported = [json.loads(line) for line in exported_path.read_text().splitlines()] + self.assertEqual(sum(item["type"] == "handoff.completion_rejected" for item in exported), 1) + recovery = [item for item in exported if item["type"] == "handoff.completion_recovered"] + self.assertEqual(len(recovery), 1) + self.assertEqual(recovery[0]["payload"], audit) + self.assertEqual(sum(item["type"] == "handoff.submitted" for item in exported), 1) + self.assert_artifacts_unchanged(artifacts) + with self.assertRaises(ConflictError): + self.service.finish(run, "accept", "Reviewed exact evidence.") + + def test_pending_finish_negative_leading_reason_round_trips_as_actual_cli(self): + run = self.normal_run("finish-negative-leading-reason") + interrupted = self.interrupt_finish(run, reason="--deferred 'quoted' assessment") + code, output, errors = self.invoke(["status", run]) + self.assertEqual((code, errors), (0, ""), output) + next_command = next(line.removeprefix("Next: ") for line in output.splitlines() if line.startswith("Next: ")) + command = shlex.split(next_command) + self.assertEqual(command, ["squad", "finish", run, "--accept", f"--reason={interrupted['decision']['reason']}"]) + code, output, errors = self.invoke(command[1:]) + self.assertEqual((code, errors), (0, ""), output) + self.assertEqual(self.service.status(run)["state"], "succeeded") + def test_pending_terminal_finish_refuses_different_intent_and_corrupt_evidence(self): run = self.normal_run("pending-finish") interrupted = self.interrupt_finish(run) From 32b94d66e57ca227deb1cfda37411570af92c055 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 16:26:01 -0700 Subject: [PATCH 184/197] WIP checkpoint: docs: record frozen R6 native audit and Council epoch acceptance boundary (2026-10-02 16:26) --- docs/plans/engineering-team/RESUME.md | 13 ++++++++++--- docs/plans/engineering-team/backlog.json | 4 ++-- 2 files changed, 12 insertions(+), 5 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index be7e598..cdb3a0e 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -119,9 +119,16 @@ timestamp/ID/hash/reason; 19 focused tests/19.207s pass, 13 negative subcases. Both follow-ups are integrated; root's final 19 focused tests pass in 7.183s. Bash/reference/diff gates pass. Final added-hunk independent review is clean; one test/13 negative subcases in 0.208s, reviewed/committed blobs equivalent. -Next freeze one full suite and exact native R6 audit, then safe install only -after acceptance. No full/native gate has started. Production remains accepted -R5/schema16. +Exact immutable R6 candidate: `7130ee7689f8cc14efacb606bcbc4c05f373298c`. +Native independent audit `44d0eecb-b5da-4c09-b1e1-c38b0d1d09ca` started +2026-10-02 23:24 UTC from accepted installed R5: one pinned gpt-6.1-sol/low +reviewer, host lead, 600s wall, no revisions/fallbacks, required diff/Bash/ +affected/reference checks, no duplicate full suite. Observe this saved run; +never start another while it is live. Next launch exactly one frozen full gate: +`env DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3.12 PYTHONPATH=plugin/core/src:test/core PYTHONWARNINGS=error::ResourceWarning python3 scripts/run-core-tests.py --failfast`. +Freeze root source/tests until it completes. Record its session/result before +any interruption. Install only after full/native acceptance. Production remains +accepted R5/schema16; this audit is its only new authorized run. R7 Council remains in isolated `r7-council`; its controlled stage flow is partial. Its source is checkpointed at `42979ba4` with 21 Council tests and 67 shared-contract tests (two optional SDK skips), Bash/reference/diff gates. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 6efbba5..1c26e83 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -47,8 +47,8 @@ {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "complete", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-closure-2026-10-02.json", "checkpoint": "Accepted dfe9976/source equivalent f87060b and installed schema16. Initial native findings repaired; exact follow-up e046a174 clean/accepted. Full 504 tests/511.241s, two optional SDK skips/no errors/failures/unraisable; nine actual installed SDK tests pass/no skips. Public complete trial/evaluate/qualify/promote/new-run/held-out regression/rollback, shared call/active deadline and all terminal projection origins pass. Safe old-schema upgrade and idempotence/drift gates pass. Failed history retained; automatic experimentation and Jev stay off."}, - {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Readiness, guided finish, human defaults, project-scoped run choices, committed-target checks and actual temporary installed normal review/fix fixtures integrated; shared catalog descendant cleanup repaired. Independent interrupted-finish P1 is repaired by atomic exact intent/current claim recovery at 22a4ed8a; 150 affected plus final 45 CLI/UX agent tests pass. Root integrated/full/native/installed acceptance remains pending. Current production remains accepted R5."}, - {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Isolated partial source 42979ba4: 21 Council tests, 67 shared tests/two optional SDK skips and Bash/reference gates pass. Controlled actual-worker reproduction confirms old schema16 client can strand a headless Council handoff; minimal epoch17 repair and old-client/upgrade gates are required. Native HTTPS isolation, reserved-launch cancellation, predeclared comparison, integrated/full/native/installed acceptance remain open."}, + {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Frozen integrated candidate7130ee7 includes durable exact guided-intent recovery, complete audited expired rejection recovery, original event verification and actual negative-leading reason CLI roundtrip. 156 agent affected tests plus final19/root19 pass; narrow independent reviews clean. Exact native audit44d0eecb started; one frozen full and installed acceptance pending. Current production remains accepted R5/schema16."}, + {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Isolated42979ba4 plus72ae4130: Council/epoch17/cancellation/comparison and affected gates pass. Actual immutable old16 client denial on17 leaves ledger unchanged and new lead succeeds. Matched/heldout actual fixture comparison is inconclusive; automatic use off, native quality/usage unknown. Native HTTPS attestation fails closed before generation. Temporary actual install upgrade, shared R6 reconciliation, full/native/installed acceptance remain open."}, {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} ], "milestones": [ From 4ae653fdc10f21cddc97bfd12e00c01e8a2252d4 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 16:39:57 -0700 Subject: [PATCH 185/197] WIP checkpoint: R6: record554-test baseline and explicit native ownership rejection; retain Council upgrade proof (2026-10-02 16:39) --- docs/plans/engineering-team/RESUME.md | 39 +++++++++----- docs/plans/engineering-team/backlog.json | 4 +- ...terminal-readiness-partial-2026-10-02.json | 54 +++++++++++++++++-- 3 files changed, 79 insertions(+), 18 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index cdb3a0e..7bf4e32 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -120,15 +120,23 @@ Both follow-ups are integrated; root's final 19 focused tests pass in 7.183s. Bash/reference/diff gates pass. Final added-hunk independent review is clean; one test/13 negative subcases in 0.208s, reviewed/committed blobs equivalent. Exact immutable R6 candidate: `7130ee7689f8cc14efacb606bcbc4c05f373298c`. -Native independent audit `44d0eecb-b5da-4c09-b1e1-c38b0d1d09ca` started -2026-10-02 23:24 UTC from accepted installed R5: one pinned gpt-6.1-sol/low -reviewer, host lead, 600s wall, no revisions/fallbacks, required diff/Bash/ -affected/reference checks, no duplicate full suite. Observe this saved run; -never start another while it is live. Next launch exactly one frozen full gate: +Native independent audit `44d0eecb-b5da-4c09-b1e1-c38b0d1d09ca` is explicitly +**rejected**, failed/version22. Verified Codex0.159.2/gpt-6.1-sol/low found P1 +`r6-probe-ownership-unavailable`: missing captured/current start identities +permit group signals and diagnostic protocol/output use. Readiness owns the +isolated fail-closed repair from7130ee7; no Council network probe ran. Diff/Bash +checks passed, but the required66-test command (89.977s/OK) was **invalidated** +by undeclared bytecode; reference was not run. The private invocation now sets +PYTHONDONTWRITEBYTECODE=1; candidate integrity was not relaxed. Exact packet +and four artifact hashes are verified and retained in R6 partial evidence. +The one frozen baseline full gate completed successfully, exec86382: `env DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3.12 PYTHONPATH=plugin/core/src:test/core PYTHONWARNINGS=error::ResourceWarning python3 scripts/run-core-tests.py --failfast`. -Freeze root source/tests until it completes. Record its session/result before -any interruption. Install only after full/native acceptance. Production remains -accepted R5/schema16; this audit is its only new authorized run. +**554 tests/645.034s**, two optional SDK skips, zero failures/errors/unraisable; +UTC/monotonic645.167s agree. RootHEAD32b94d6 source/tests/scripts are equivalent +to7130ee7. This is a baseline pass, not proof of the later ownership repair. +No full suite/native run remains active. Next integrate the ownership repair, +run affected regressions and a narrow exact follow-up audit; install only after +final frozen/native acceptance. Production remains accepted R5/schema16. R7 Council remains in isolated `r7-council`; its controlled stage flow is partial. Its source is checkpointed at `42979ba4` with 21 Council tests and 67 shared-contract tests (two optional SDK skips), Bash/reference/diff gates. @@ -147,10 +155,17 @@ quality/escaped-defect/rework/quota/host-usage observations remain unknown, automatic use off. Its 25 Council tests/90.724s, 35 shared migration tests/ 54.469s, 12 handoff-store tests/4.528s and 10 supervisor/crash tests/1.819s pass. Native HTTPS attestation fails closed before generation. The Council agent -owns actual temporary immutable16-to17 install-upgrade proof; readiness owns -immutable-old16 claim denial and one bounded Unix mDNS resolver-socket diagnostic. -Root owns shared-hunk/formatter reconciliation after R6 acceptance. No production -edit or new native Council generation has occurred. Preserve attached worktrees. +completed the actual temporary immutable16-to17 install-upgrade proof at +`eee6db738a9996ea432e3ca2bf2347b005ecd74a`: active and recoverable16 both defer +without selector/launcher/schema changes; safe cancel/reconcile permits17, +then immutable long-lived/fresh old16 clients deny Council mutations. Two +focused tests/7.976s, Bash/reference/diff pass; no production update occurred. +Readiness independently verified old16 denial on17 and the actual fourth lead +succeeding. Its one bounded Unix mDNS resolver-socket diagnostic is paused for +the P1 ownership repair. Council agent now owns shared-hunk/formatter and R6 +authority reconciliation on isolated `codex/council-integration` in the former +r6-terminal worktree; root remains on codex/engineering-team. No native Council +generation occurred. Preserve all original checkpoints/attached worktrees. Desktop control worked for scoped inspection. Claude's local Code tab selected only this DevSquad project on `codex/engineering-team`, with an empty prompt; no proof diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 1c26e83..7846811 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -47,8 +47,8 @@ {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "complete", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-closure-2026-10-02.json", "checkpoint": "Accepted dfe9976/source equivalent f87060b and installed schema16. Initial native findings repaired; exact follow-up e046a174 clean/accepted. Full 504 tests/511.241s, two optional SDK skips/no errors/failures/unraisable; nine actual installed SDK tests pass/no skips. Public complete trial/evaluate/qualify/promote/new-run/held-out regression/rollback, shared call/active deadline and all terminal projection origins pass. Safe old-schema upgrade and idempotence/drift gates pass. Failed history retained; automatic experimentation and Jev stay off."}, - {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Frozen integrated candidate7130ee7 includes durable exact guided-intent recovery, complete audited expired rejection recovery, original event verification and actual negative-leading reason CLI roundtrip. 156 agent affected tests plus final19/root19 pass; narrow independent reviews clean. Exact native audit44d0eecb started; one frozen full and installed acceptance pending. Current production remains accepted R5/schema16."}, - {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Isolated42979ba4 plus72ae4130: Council/epoch17/cancellation/comparison and affected gates pass. Actual immutable old16 client denial on17 leaves ledger unchanged and new lead succeeds. Matched/heldout actual fixture comparison is inconclusive; automatic use off, native quality/usage unknown. Native HTTPS attestation fails closed before generation. Temporary actual install upgrade, shared R6 reconciliation, full/native/installed acceptance remain open."}, + {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Baseline7130ee7/root32b94 passes554 full tests/645.034s, two optional SDK skips/no errors/failures/unraisable. Exact native audit44d0eecb explicitly rejected/failed22: unavailable probe ownership P1 and undeclared-bytecode invalidated check (66 tests ranOK, not accepted). Isolated ownership repair pending; corrected no-bytecode invocation preserves integrity. Guided finish recovery repairs and narrow reviews remain verified. No full/native process active; final repaired/full/native/installed acceptance pending. Production remains accepted R5/schema16."}, + {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Isolated42979ba4/72ae4130/eee6db7: Council/epoch17/cancellation/comparison and actual immutable16-to17 temporary install proof pass. Active/recoverable16 defer unchanged, cancel/reconcile safely permits17; actual long-lived/fresh old16 clients reject Council mutation, ledger unchanged, new lead succeeds. Matched/heldout fixture comparison inconclusive; automatic use off, native quality/usage unknown. Native HTTPS attestation fails closed before generation; diagnostic paused for R6 ownership P1. Shared R6 authority/formatter reconciliation isolated on codex/council-integration; integrated full/native/installed acceptance remain open."}, {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} ], "milestones": [ diff --git a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json index 2011419..35b905e 100644 --- a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json @@ -2,7 +2,7 @@ "schema_version": 1, "work_package": "R6", "recorded_on": "2026-10-02", - "status": "recovery_repairs_integrated_full_native_pending", + "status": "baseline_full_passed_native_rejected_ownership_repair_pending", "base_revision": "8f2c1c6a7d9edfe7cc3d78ecbc59ce105be636b2", "integrated_checkpoints": [ "dd709804e70e0791614319a9ae5c2597768d7318", @@ -60,8 +60,53 @@ "First expanded repair-agent gate failed a stale isolated bf3 current-schema expectation of 15; root already expected 16 after dfe9976. Changed only the current-schema assertion to SUPPORTED_SCHEMA_VERSION and retained the historical schema-four fixture checks; not a current root runtime failure", "Independent 22a4ed8a follow-up: ten existing repair tests pass in 26.811s, but a lease expiring after claim acquisition and before submission creates a durable expired_claim rejection. A fresh own retry advances fence 2 to 3/version 16 to 17 yet the stable decision replays that old rejection indefinitely. Follow-up must preserve complete immutable rejection history while recovering only exact fresh guided authority", "Independent 22a4ed8a command-copyability edge: saved reason --deferred is printed as --reason --deferred and argparse rejects it. Fixed by 31a8a68b attached shell-quoted --reason= and actual command round-trip", - "Follow-up red reproduction: two tests/3.845s, one failure and one error. Initial audit implementation reused a run_version for two events and failed uniqueness; consecutive atomic recovery/submitted versions repaired it before green checkpoint gates" + "Follow-up red reproduction: two tests/3.845s, one failure and one error. Initial audit implementation reused a run_version for two events and failed uniqueness; consecutive atomic recovery/submitted versions repaired it before green checkpoint gates", + "Exact native audit44d0eecb found a high process-ownership defect: start_identity=None/current_identity=None/returncode=None permits group signaling; diagnostic callers use output/protocol without confirmed captured identity. Candidate explicitly rejected, failed/version22; isolated repair pending", + "Native required check ran66 tests/89.977s/OK but created nine undeclared cpython-314 bytecode files. Integrity correctly invalidated the check; subsequent reference not run. Corrected private invocation sets PYTHONDONTWRITEBYTECODE=1 without relaxing candidate integrity" ], + "baseline_full_gate": { + "target_revision": "7130ee7689f8cc14efacb606bcbc4c05f373298c", + "frozen_root_head": "32b94d66e57ca227deb1cfda37411570af92c055", + "source_tests_scripts_equivalent": true, + "exec_session": 86382, + "command": "env DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3.12 PYTHONPATH=plugin/core/src:test/core PYTHONWARNINGS=error::ResourceWarning python3 scripts/run-core-tests.py --failfast", + "tests": 554, + "seconds": 645.034, + "monotonic_seconds": 645.1674028339985, + "utc_seconds": 645.167055, + "optional_sdk_skips": 2, + "failures": 0, + "errors": 0, + "unraisable": [], + "scope": "Baseline only; does not prove the pending ownership repair or Council integration" + }, + "native_audit": { + "run_id": "44d0eecb-b5da-4c09-b1e1-c38b0d1d09ca", + "target_revision": "7130ee7689f8cc14efacb606bcbc4c05f373298c", + "candidate_sha256": "2751f6b40a2125a83ee10bc6047beee378ec617bff0817c0e26666fc76d00ee6", + "state": "failed", + "version": 22, + "host_disposition": "reject", + "accept_allowed": false, + "review_verdict": "findings", + "finding": "r6-probe-ownership-unavailable", + "severity": "high", + "identity": {"harness":"codex","harness_version":"codex-cli 0.159.2","model_id":"gpt-6.1-sol","effort":"low","verification":"verified","permission_policy":"read_only"}, + "usage": {"source":"native_reported","input_tokens":276520,"output_tokens":2246,"total_tokens":278766}, + "usage_limitation": "Not a subscription invoice or an estimate of Plus five-hour windows", + "artifact_hashes_verified": [ + "c82d6c3d20b6b93ba93a3e0d1f6cac22bb6d5a80d453e47432dbd40120c0dd79", + "cdfe7b8dd8023747ac76539dd835f5b5b0fe3e71f833c31593287dc77b78e39c", + "8c911a4fa0be95aeeba341a75978dcb109f977b7bab91462d4e9150150e4e721", + "3ed7a9a766d3f19c0ff73c956dbb196ee550c15d86f58bf38827cb75820eca51" + ], + "checks": [ + {"id":"candidate-diff-check","status":"passed","integrity":"verified","duration_ms":20}, + {"id":"detected-tests","status":"passed","integrity":"verified","duration_ms":16912}, + {"id":"user-check-1","status":"invalidated","integrity":"violated","duration_ms":90564,"tests_ran":66,"test_seconds":89.977,"test_runner_result":"OK","reason":"check:undeclared_inputs_changed"}, + {"id":"user-check-2","status":"not_run","reason":"prior_check_invalidated_candidate"} + ] + }, "independent_followup": {"revision": "31a8a68b", "result": "clean_bounded_scope", "tests": 6, "seconds": 9.120, "source_blobs_before_after": "unchanged", "full_five_file_diff_sha256": "b4c742fa400932d216d698e483af40abaa733c741325aebd3eaaf56776accf40", "final_original_event_hunk_review": {"revision": "e7e9042c", "result": "clean_bounded_scope", "tests": 1, "negative_subcases": 13, "seconds": 0.208, "source_before_after_committed": "equivalent", "two_file_diff_sha256": "39ba13f6e5b0cf06da5c55d619d7bd0b2a3684780129d16f8493640736b8cc65"}}, "fresh_install_scope": { "providers": "offline native CLI protocol fixtures only", @@ -83,8 +128,9 @@ "unrelated_servers_or_settings": "not changed" }, "remaining_gates": [ - "Frozen full gate", - "Independent exact R6 review and accepted immutable checkpoint", + "Fail-closed unavailable process ownership regression/repair and independent exact follow-up", + "Final frozen integration full gate; baseline554 pass is not repaired-source acceptance", + "Independent exact R6 acceptance and immutable checkpoint", "Safe installed update, non-generating live readiness, idempotence/drift and actual installed SDK tests", "Separate Claude Code-tab if accessible, updated Antigravity CLI and R7/R8 gates; Antigravity IDE not required by user clarification" ] From 8e593494b2b26d3e752267ea0656f6bef1161238 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 16:48:41 -0700 Subject: [PATCH 186/197] WIP checkpoint: docs: shorten resume history and distinguish installed R5 from historical G4 (2026-10-02 16:48) --- docs/plans/engineering-team/RESUME.md | 81 ++++++--------------------- 1 file changed, 16 insertions(+), 65 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 7bf4e32..ab2ea49 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,71 +6,22 @@ remain recoverable in Git; detailed receipts and failed gates stay in evidence. ## Current checkpoint — October 2, 2026 -R3c is now **accepted** at immutable `39b95f1`: native independent audit is -clean; mandatory diff/Bash/51 focused tests pass with unchanged integrity. -The R3 source blobs exactly match the prior accepted 477-test full candidate. -See `evidence/R3-closure-2026-10-02.json` for the requirement/proof matrix. -R4 is now **accepted and installed** at source `913abc1`: +Accepted work must not be restarted or re-tested unchanged: -- Normal entry has private account/config/binary/version-scoped complete - last-good catalogs, 24h TTL, one OS refresh owner, bounded pagination and - two-minute failure backoff. Default hints do not promote or invalidate aliases. -- Native typed quota observations share one opaque **account-only** reservation - pool across discovery configurations. Incompatible qualified contexts block; - pins remain trials. Failed queries retain still-fresh known exhaustion. -- 102 affected tests in 17.610s; **484 full tests in 458.607s**, two optional SDK - skips, zero failures/errors/unraisable. Source was frozen for the full gate. -- Initial R4 audit `9a12f6f6…` rejected a high shared-pool partition bug. - Exact repair audit **`f45c723b-3067-473f-9349-f1012f5548e7`** is clean, - succeeded/22; diff/Bash/30 targeted tests pass, one optional wheel-environment - skip, all integrity hashes unchanged. Initial interrupted failed full gate - and the fixed CLI expectations remain truthfully recorded in partial evidence. -- Installed normal dry-run selects gpt-6.1-sol/low as a bounded trial, no run - creation; reinstall no-op, drift false, pip check pass, four registrations - matching. **Nine actual installed SDK transport tests pass in 2.743s, no skips.** +- R3 `39b95f1`: exact native review clean, focused/public/historical evidence + and source-equivalent prior477 full pass. [Closure](evidence/R3-closure-2026-10-02.json). +- R4 `913abc1`: normal catalogs, account-wide quota fences and guarded aliases; + repaired exact native audit accepted,484 full tests, installed SDK9 pass. + [Closure](evidence/R4-closure-2026-10-02.json). +- **Current installed R5 `dfe9976`, schema16**: public outcomes/trials and + shared experiment budgets; repair audit `e046a174-55ce-4496-973e-2fc97c68a493` + accepted,504 full tests/511.241s, installed SDK9/2.674s, no drift and four + registrations matching. [Closure](evidence/R5-closure-2026-10-02.json). -Closure matrix: `evidence/R4-closure-2026-10-02.json`. No native audit, full -suite or nonterminal production run remains active. Earlier accepted native -Claude/Grok/Gemini proofs below remain historical; do not repeat unchanged ones. -Antigravity externally updated to **1.2.14**: current doctor correctly labels -its adapter unverified; the old 1.2.13 live status receipt is not a new-version -proof. Recheck this during R6/R8 without broad MCP listings or UI bypass. - -**R5 is now accepted and installed** at source **dfe9976**, schema **16**. -Closure matrix: `evidence/R5-closure-2026-10-02.json`. The new -projection outbox opts in new public runs without rewriting legacy history. -Terminal transitions and targeted status/result/report/evaluation replay -generate one final outcome; failures, repairs, findings, fenced lead -attestations and late corrections remain separate. Corrupt pending evidence -does not block unrelated runs. The explicit public `trial` / `trial_start` -path now uses the schema-14 assignment fence; the fixture no longer patches -preparation or manually imports finals. Experiment reservations share a -durable all-attempt call cap and a declaration-time wall deadline; automatic -experimentation stays off. -Public concurrent-budget and promotion → new-run binding → held-out regression -→ rollback tests pass. See `evidence/R5-public-integration-partial-2026-10-02.json`. - -Initial native R5 audit rejected two real defects. Project projection now -precedes the consistent proposal read; launch/active-worker deadlines include -the shared experiment deadline. Exact repair follow-up -**e046a174-55ce-4496-973e-2fc97c68a493** is clean/accepted, succeeded/22; -diff/Bash/13 public tests/reference checks pass with unchanged integrity. -The 54 repaired-path and 66 integrity/outcome/review/delivery tests pass. -Final frozen **504-test full gate passes in 511.241s**, two optional SDK skips, -zero errors/failures/unraisable, UTC/monotonic ~511.32s. Later revisions change -only test observations/schema expectations and documents, not reviewed code. - -Safe offline installation selected `0.1.0-py31214-90f1e87fb9ac-mcp-a26bc88afbef`; -zero active production runs, lazy migration to 16, zero retroactive objective -jobs. Reinstall no-op, payload drift false, pip check pass, four registrations -matching. **Nine actual installed SDK tests pass in 2.674s without skips.** -Installed normal review dry-run still resolves gpt-6.1-sol/low bounded trial, -without creating a run. Full installer tests include actual old-schema active -deferral, cancellation/recovery and migration to current16. Failed red tests, -rejected audit and interrupted/failed full gates remain in partial evidence; -do not repeat accepted proofs. No test/native process is active here. -Next: integrate/test R6 terminal/readiness, R7/C1 and final R8 acceptance; -whole-plan completion is not claimed. +Failed/red/interrupted gates remain in each package's partial evidence. The +remaining delivery is R6 → R7/C1 → final R8 acceptance, not architecture work. +Antigravity is **agy CLI**, now1.2.14; its final-release operation is pending. +The older1.2.13 receipt is historical, not proof of the updated CLI/adapter. R6 terminal/readiness checkpoints `dd709804`, `79f9b767`, `b785d040`, `d0eec55c`, shared helper `4a6bc15d` and catalog reuse `03124c79` are now @@ -178,8 +129,8 @@ the accepted final release. Unrelated servers/settings remain untouched. Workspace: /Users/Dikshant/Desktop/Projects/devsquad. Branch: `codex/engineering-team`; never restart this build from main. -Source runtime repairs and the accepted G4 candidate below are **integrated -and installed at `6d2e0ba`**; all five blob hashes +Historical runtime repairs and the accepted G4 candidate below were integrated +and installed at `6d2e0ba` (superseded by the current R5 release); all five blob hashes exactly match the accepted candidate. Local affected gate: 27 tests in 18.940s, plus 227 Bash assertions, generated-reference and diff checks passed. From 8242eff68b24538aacc1b40b67b2e88e96b04d52 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 16:54:50 -0700 Subject: [PATCH 187/197] WIP checkpoint: R6: integrate fail-closed probe ownership and retained Python3.12 cleanup anchor (2026-10-02 16:54) --- docs/plans/engineering-team/RESUME.md | 19 +- docs/plans/engineering-team/backlog.json | 2 +- ...terminal-readiness-partial-2026-10-02.json | 17 +- plugin/core/src/devsquad/diagnostics.py | 16 +- plugin/core/src/devsquad/probe_process.py | 144 +++++++- test/core/test_diagnostics.py | 348 ++++++++++++++++++ test/core/test_task_entry.py | 13 +- 7 files changed, 533 insertions(+), 26 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index ab2ea49..3a4fc28 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -74,8 +74,18 @@ Exact immutable R6 candidate: `7130ee7689f8cc14efacb606bcbc4c05f373298c`. Native independent audit `44d0eecb-b5da-4c09-b1e1-c38b0d1d09ca` is explicitly **rejected**, failed/version22. Verified Codex0.159.2/gpt-6.1-sol/low found P1 `r6-probe-ownership-unavailable`: missing captured/current start identities -permit group signals and diagnostic protocol/output use. Readiness owns the -isolated fail-closed repair from7130ee7; no Council network probe ran. Diff/Bash +permit group signals and diagnostic protocol/output use. Repair `093ed0f397289bffe9ea249c10796cb436027f00` +is now integrated: missing capture blocks reads/peer/send and all group signals; +only kernel-confirmed direct children are stopped/reaped. Surviving descendants +without authority remain cleanup-unconfirmed. Known missing-current identity +requires a retained unreaped waitid or exact bounded zombie PID/PPID/PGID anchor +under the Popen reap lock; fake/reaped/live-ps/wrong-parent/wrong-group rows fail +closed. EOF preserves the child PID through cleanup. No ABI additions; helper +requires exclusive Popen reaping, never an external waitpid/SIGCHLD reaper. +Six red baseline regressions/1.216s; accepted Python3.12 strict36/11.250s and +Python3.14 strict36/11.364s pass; root seven/1.259s with exact matching blobs, +Bash/reference/diff green. No Council network +probe ran. Diff/Bash checks passed, but the required66-test command (89.977s/OK) was **invalidated** by undeclared bytecode; reference was not run. The private invocation now sets PYTHONDONTWRITEBYTECODE=1; candidate integrity was not relaxed. Exact packet @@ -85,8 +95,9 @@ The one frozen baseline full gate completed successfully, exec86382: **554 tests/645.034s**, two optional SDK skips, zero failures/errors/unraisable; UTC/monotonic645.167s agree. RootHEAD32b94d6 source/tests/scripts are equivalent to7130ee7. This is a baseline pass, not proof of the later ownership repair. -No full suite/native run remains active. Next integrate the ownership repair, -run affected regressions and a narrow exact follow-up audit; install only after +No full suite/native run remains active. Next save this ownership integration +after root affected tests/Bash, then run one narrow exact follow-up audit and +the prepared bounded nongenerating resolver-socket diagnostic. Install only after final frozen/native acceptance. Production remains accepted R5/schema16. R7 Council remains in isolated `r7-council`; its controlled stage flow is partial. Its source is checkpointed at `42979ba4` with 21 Council tests and diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 7846811..0448686 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -47,7 +47,7 @@ {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "complete", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-closure-2026-10-02.json", "checkpoint": "Accepted dfe9976/source equivalent f87060b and installed schema16. Initial native findings repaired; exact follow-up e046a174 clean/accepted. Full 504 tests/511.241s, two optional SDK skips/no errors/failures/unraisable; nine actual installed SDK tests pass/no skips. Public complete trial/evaluate/qualify/promote/new-run/held-out regression/rollback, shared call/active deadline and all terminal projection origins pass. Safe old-schema upgrade and idempotence/drift gates pass. Failed history retained; automatic experimentation and Jev stay off."}, - {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Baseline7130ee7/root32b94 passes554 full tests/645.034s, two optional SDK skips/no errors/failures/unraisable. Exact native audit44d0eecb explicitly rejected/failed22: unavailable probe ownership P1 and undeclared-bytecode invalidated check (66 tests ranOK, not accepted). Isolated ownership repair pending; corrected no-bytecode invocation preserves integrity. Guided finish recovery repairs and narrow reviews remain verified. No full/native process active; final repaired/full/native/installed acceptance pending. Production remains accepted R5/schema16."}, + {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Baseline7130ee7/root32b94 passes554 full tests/645.034s, two optional SDK skips/no errors/failures/unraisable. Exact native audit44d0eecb explicitly rejected/failed22: unavailable probe ownership P1 and bytecode-invalidated check (66 tests ranOK, not accepted). Integrated repair093ed0f: strict36 tests on accepted3.12/11.250s and3.14/11.364s; root7/1.259s, matching source blobs, Bash/reference/diff pass. No captured identity means no protocol/output/group signals; kernel-confirmed direct-child fallback and retained exact zombie anchor preserve cleanup safely. Corrected no-bytecode invocation preserves integrity. Narrow native follow-up and final integrated/full/installed acceptance pending. Production remains accepted R5/schema16."}, {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Isolated42979ba4/72ae4130/eee6db7: Council/epoch17/cancellation/comparison and actual immutable16-to17 temporary install proof pass. Active/recoverable16 defer unchanged, cancel/reconcile safely permits17; actual long-lived/fresh old16 clients reject Council mutation, ledger unchanged, new lead succeeds. Matched/heldout fixture comparison inconclusive; automatic use off, native quality/usage unknown. Native HTTPS attestation fails closed before generation; diagnostic paused for R6 ownership P1. Shared R6 authority/formatter reconciliation isolated on codex/council-integration; integrated full/native/installed acceptance remain open."}, {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} ], diff --git a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json index 35b905e..2716976 100644 --- a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json @@ -2,7 +2,7 @@ "schema_version": 1, "work_package": "R6", "recorded_on": "2026-10-02", - "status": "baseline_full_passed_native_rejected_ownership_repair_pending", + "status": "ownership_repair_integrated_followup_final_full_install_pending", "base_revision": "8f2c1c6a7d9edfe7cc3d78ecbc59ce105be636b2", "integrated_checkpoints": [ "dd709804e70e0791614319a9ae5c2597768d7318", @@ -13,7 +13,8 @@ "03124c79ba53cb1b12296bb554024605e4cadbc7", "22a4ed8ac85b95861f4ba9725ca729388b7a9f49", "31a8a68b59db3d0af136c5c18cb67fe5088c02b0", - "e7e9042cbf4d1416209cf39e0269f7fa85479f4c" + "e7e9042cbf4d1416209cf39e0269f7fa85479f4c", + "093ed0f397289bffe9ea249c10796cb436027f00" ], "contract": "SOL-REVIEW-FOLLOWUP R6/G3/G4; unchanged low-level MCP envelopes, fenced host disposition, Bash 3.2 and optional jq", "implemented": [ @@ -38,14 +39,17 @@ {"revision": "22a4ed8a", "tests": 150, "seconds": 180.750, "failures": 0, "errors": 0, "scope": "Affected repair gate before final command-copyability edit"}, {"revision": "22a4ed8a", "tests": 45, "seconds": 48.316, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Final CLI/terminal repair checkpoint"}, {"revision": "31a8a68b", "tests": 156, "seconds": 350.005, "skips": 0, "failures": 0, "errors": 0, "resource_warnings": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Expiry rejection and command round-trip repair, before final original-rejection-event crosscheck"}, - {"revision": "e7e9042c", "tests": 19, "seconds": 19.207, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Final original-rejection-event crosscheck; 13 negative audit-corruption subcases"} + {"revision": "e7e9042c", "tests": 19, "seconds": 19.207, "failures": 0, "errors": 0, "bash_assertions": 227, "generated_reference": "pass", "diff_check": "pass", "scope": "Final original-rejection-event crosscheck; 13 negative audit-corruption subcases"}, + {"revision":"093ed0f397289bffe9ea249c10796cb436027f00","tests":36,"seconds":11.250,"python":"accepted immutable3.12","resource_warnings":0,"failures":0,"errors":0,"bash_assertions":227,"generated_reference":"pass","diff_check":"pass","scope":"Unavailable capture/current ownership, kernel direct-child-only cleanup, retained exited zombie and TERM-ignoring descendant/EOF cleanup; exclusive Popen-reaping contract"}, + {"revision":"093ed0f397289bffe9ea249c10796cb436027f00","tests":36,"seconds":11.364,"python":"3.14","resource_warnings":0,"failures":0,"errors":0,"scope":"Same affected ownership and native-discovery regression set"} ], "root_affected_gates": [ {"command": "python3 -m unittest test_diagnostics test_terminal_ux test_installed_usability test_cli test_task_entry test_mcp test_handoff_store", "tests": 114, "seconds": 65.066, "optional_sdk_skips": 2, "failures": 0, "errors": 0, "scope": "Integrated first five checkpoints, before final catalog cleanup reuse"}, {"command": "python3 -m unittest test_delivery_workflow test_delivery_identity_runtime test_handoff_service", "tests": 32, "seconds": 66.877, "failures": 0, "errors": 0}, {"command": "python3 -m unittest test_task_entry test_installed_usability", "tests": 27, "seconds": 25.389, "skips": 0, "failures": 0, "errors": 0, "scope": "All six integrated checkpoints including final catalog cleanup reuse"}, {"command": "python3 -m unittest test_terminal_ux test_cli test_handoff_store test_handoff_service", "tests": 61, "seconds": 55.963, "failures": 0, "errors": 0, "scope": "Integrated initial interrupted-finish repair and current-schema assertion; ResourceWarning strict"}, - {"command": "python3 -m unittest test_handoff_store test_terminal_ux.TerminalUxTest.test_finish_retries_authoritative_expiry_rejection_with_exact_fresh_claim test_terminal_ux.TerminalUxTest.test_pending_finish_negative_leading_reason_round_trips_as_actual_cli", "tests": 19, "seconds": 7.183, "failures": 0, "errors": 0, "scope": "All recovery follow-ups including final exact original rejection audit; ResourceWarning strict"} + {"command": "python3 -m unittest test_handoff_store test_terminal_ux.TerminalUxTest.test_finish_retries_authoritative_expiry_rejection_with_exact_fresh_claim test_terminal_ux.TerminalUxTest.test_pending_finish_negative_leading_reason_round_trips_as_actual_cli", "tests": 19, "seconds": 7.183, "failures": 0, "errors": 0, "scope": "All recovery follow-ups including final exact original rejection audit; ResourceWarning strict"}, + {"tests":7,"seconds":1.259,"python":"accepted immutable3.12","resource_warnings":0,"failures":0,"errors":0,"source_blobs_equivalent_to":"093ed0f397289bffe9ea249c10796cb436027f00","scope":"Root independent capture/currentNone, no protocol/false readiness, direct-child absence, bounded zombie authority and EOF descendant checks"} ], "failure_history": [ "Readiness red baseline: nine tests, three failures and 17 subtest errors before implementation", @@ -61,8 +65,9 @@ "Independent 22a4ed8a follow-up: ten existing repair tests pass in 26.811s, but a lease expiring after claim acquisition and before submission creates a durable expired_claim rejection. A fresh own retry advances fence 2 to 3/version 16 to 17 yet the stable decision replays that old rejection indefinitely. Follow-up must preserve complete immutable rejection history while recovering only exact fresh guided authority", "Independent 22a4ed8a command-copyability edge: saved reason --deferred is printed as --reason --deferred and argparse rejects it. Fixed by 31a8a68b attached shell-quoted --reason= and actual command round-trip", "Follow-up red reproduction: two tests/3.845s, one failure and one error. Initial audit implementation reused a run_version for two events and failed uniqueness; consecutive atomic recovery/submitted versions repaired it before green checkpoint gates", - "Exact native audit44d0eecb found a high process-ownership defect: start_identity=None/current_identity=None/returncode=None permits group signaling; diagnostic callers use output/protocol without confirmed captured identity. Candidate explicitly rejected, failed/version22; isolated repair pending", - "Native required check ran66 tests/89.977s/OK but created nine undeclared cpython-314 bytecode files. Integrity correctly invalidated the check; subsequent reference not run. Corrected private invocation sets PYTHONDONTWRITEBYTECODE=1 without relaxing candidate integrity" + "Exact native audit44d0eecb found a high process-ownership defect: start_identity=None/current_identity=None/returncode=None permits group signaling; diagnostic callers use output/protocol without confirmed captured identity. Candidate explicitly rejected, failed/version22; repair093ed0f now integrated, exact follow-up pending", + "Native required check ran66 tests/89.977s/OK but created nine undeclared cpython-314 bytecode files. Integrity correctly invalidated the check; subsequent reference not run. Corrected private invocation sets PYTHONDONTWRITEBYTECODE=1 without relaxing candidate integrity", + "Ownership repair red: six regressions/1.216s expose baseline signal/read/send/false-authentication behavior. First conservative macOS3.12 repair refused known exited-parent descendant cleanup because waitid is unavailable; fixture child was removed. Exact bounded retained zombie-child anchor and EOF-before-reap contract restored this required cleanup without unsafe missing-capture signals" ], "baseline_full_gate": { "target_revision": "7130ee7689f8cc14efacb606bcbc4c05f373298c", diff --git a/plugin/core/src/devsquad/diagnostics.py b/plugin/core/src/devsquad/diagnostics.py index 128a7c4..580ddc5 100644 --- a/plugin/core/src/devsquad/diagnostics.py +++ b/plugin/core/src/devsquad/diagnostics.py @@ -53,8 +53,11 @@ def _probe_output( argv, cwd=project, env=environment, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, start_new_session=True, ) - start_identity = capture_probe_identity(process) + start_identity = None try: + start_identity = capture_probe_identity(process) + if start_identity is None: + raise ContractError("diagnostic process ownership is unavailable") assert process.stdout is not None chunks = bytearray() with selectors.DefaultSelector() as selector: @@ -69,12 +72,15 @@ def _probe_output( chunks.extend(chunk) if len(chunks) > MAX_PROBE_BYTES: raise ContractError("diagnostic probe exceeds byte bound") - remaining = deadline - time.monotonic() - if remaining <= 0: + if deadline - time.monotonic() <= 0: raise TimeoutError("diagnostic probe timed out") - return process.wait(timeout=remaining), chunks.decode("utf-8") + output = chunks.decode("utf-8") finally: + # Retain the direct child's PID until group cleanup is confirmed; + # waiting here first would discard the exited-parent ownership anchor. _close_probe(process, start_identity=start_identity) + assert process.returncode is not None + return process.returncode, output def _resolve_adapter( @@ -170,6 +176,8 @@ def _codex_auth(binary: str, *, project: Path, environment: dict[str, str]) -> d bufsize=1, start_new_session=True, ) start_identity = capture_probe_identity(process) + if start_identity is None: + raise ContractError("diagnostic process ownership is unavailable") assert process.stdin is not None and process.stdout is not None peer = JsonLinePeer(process.stdout, process.stdin, max_frame_bytes=MAX_PROBE_BYTES) peer.send(initialize_request(1)) diff --git a/plugin/core/src/devsquad/probe_process.py b/plugin/core/src/devsquad/probe_process.py index 72f81dc..77ef18a 100644 --- a/plugin/core/src/devsquad/probe_process.py +++ b/plugin/core/src/devsquad/probe_process.py @@ -2,6 +2,7 @@ from __future__ import annotations +from contextlib import contextmanager import getpass import os from pathlib import Path @@ -15,6 +16,11 @@ PROBE_CLEANUP_SECONDS = 2 PROBE_TERM_GRACE_SECONDS = 0.25 +_POPEN_TYPE = subprocess.Popen + + +class _ProbeOwnershipUnavailable(ContractError): + pass def subscription_environment(home: Path | None = None) -> dict[str, str]: @@ -59,34 +65,150 @@ def _probe_group_exists(pgid: int, *, deadline: float) -> bool: return False +@contextmanager +def _reap_guard(process: subprocess.Popen[Any], *, deadline: float, locked: bool = False): + if locked or not isinstance(process, _POPEN_TYPE): + yield + return + remaining = deadline - time.monotonic() + if remaining <= 0 or not process._waitpid_lock.acquire(timeout=remaining): + raise ContractError("diagnostic cleanup deadline expired") + try: + yield + finally: + process._waitpid_lock.release() + + +def _retained_child_anchor( + process: subprocess.Popen[Any], *, deadline: float, locked: bool = False, +) -> bool: + """Verify an unreaped child without releasing its PID for reuse. + + Callers exclusively reap this Popen through wait/poll, never a separate + os.waitpid/SIGCHLD reaper. Under its reap lock, waitid(WNOWAIT) or an exact + kernel zombie-child row retains the PID through any final group signal. + A live ps row or Popen.returncode alone is not this authority. + """ + if not isinstance(process, _POPEN_TYPE): + return False + with _reap_guard(process, deadline=deadline, locked=locked): + if process.returncode is not None: + return False + waitid = getattr(os, "waitid", None) + if callable(waitid): + try: + result = waitid(os.P_PID, process.pid, os.WEXITED | os.WNOHANG | os.WNOWAIT) + except (OSError, AttributeError): + return False + return result is None or result.si_pid == process.pid + # Python 3.12 on macOS lacks waitid. Only its unreaped zombie child + # in our original new-session group can substitute; never a live row. + remaining = deadline - time.monotonic() + if remaining <= 0: + raise ContractError("diagnostic cleanup deadline expired") + try: + result = subprocess.run( + ["/bin/ps", "-p", str(process.pid), "-o", "pid=,ppid=,pgid=,stat="], + text=True, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, + env=subscription_environment(), timeout=min(0.5, remaining), check=False, + ) + except (OSError, subprocess.TimeoutExpired): + return False + if result.returncode != 0 or len(result.stdout) > 1024: + return False + rows = result.stdout.splitlines() + if len(rows) != 1: + return False + fields = rows[0].split() + return (len(fields) == 4 and all(field.isascii() and field.isdigit() for field in fields[:3]) + and tuple(map(int, fields[:3])) == (process.pid, os.getpid(), process.pid) + and fields[3].startswith("Z")) + + +def _close_direct_child(process: subprocess.Popen[Any], *, deadline: float) -> None: + """Stop only a kernel-confirmed live child, never its unowned group. + + WNOHANG verifies child authority under Popen's reap lock before each + direct signal. An exited child is reaped, not signaled. No waitid or + platform-specific ABI is needed for this fail-closed fallback. + """ + if not isinstance(process, _POPEN_TYPE): + raise ContractError("diagnostic direct-child ownership is unavailable") + + def child_live(value: int | None = None) -> bool: + if time.monotonic() >= deadline: + raise ContractError("diagnostic cleanup deadline expired") + with _reap_guard(process, deadline=deadline): + if process.returncode is not None: + return False + try: + waited, status = os.waitpid(process.pid, os.WNOHANG) + except OSError as exc: + raise ContractError("diagnostic direct-child ownership is unavailable") from exc + if waited == process.pid: + process._handle_exitstatus(status) + return False + if waited != 0: + raise ContractError("diagnostic direct-child ownership is unavailable") + if value is not None: + try: + os.kill(process.pid, value) + except ProcessLookupError: + pass + return True + + if child_live(signal.SIGTERM): + grace = min(deadline, time.monotonic() + PROBE_TERM_GRACE_SECONDS) + while child_live() and time.monotonic() < grace: + time.sleep(min(0.02, max(0, grace - time.monotonic()))) + if child_live(signal.SIGKILL): + while child_live(): + time.sleep(min(0.02, max(0, deadline - time.monotonic()))) + + def close_probe(process: subprocess.Popen[Any], *, start_identity: str | None) -> None: """Stop the owned group, confirm no live members, reap and close streams. The process must have been spawned by the caller with a new session and - its identity captured before protocol use. Inspection/ownership failures - raise rather than pretending cleanup or readiness was confirmed. + its identity captured before protocol use. Callers retain exclusive Popen + reaping until this function; no separate waitpid/SIGCHLD reaper may release + the child PID. Inspection/ownership failures + raise rather than pretending cleanup or readiness was confirmed. Missing + captured identity permits no group signals: only independently verified + direct-child stop/reap, followed by group absence or an unconfirmed error. """ deadline = time.monotonic() + PROBE_CLEANUP_SECONDS - def owned_group_exists() -> bool: + def owned_group_exists(*, locked: bool = False) -> bool: if not _probe_group_exists(process.pid, deadline=deadline): return False observed = process_start_identity(process.pid) if observed is not None and observed != start_identity: raise ContractError("diagnostic process identity changed before cleanup") - if observed is None and start_identity is None and process.returncode is not None: - raise ContractError("diagnostic process ownership unavailable") + if observed is None and not _retained_child_anchor(process, deadline=deadline, locked=locked): + raise _ProbeOwnershipUnavailable("diagnostic process ownership unavailable") return True def signal_group(value: int) -> None: # Recheck immediately before each signal. A recycled leader PID is # never authority to signal a new group. - if owned_group_exists(): - try: - os.killpg(process.pid, value) - except ProcessLookupError: - pass + # The lock retains any unreaped child anchor through the signal, even + # if another caller tries Popen.wait/poll during cleanup. + with _reap_guard(process, deadline=deadline): + if owned_group_exists(locked=True): + try: + os.killpg(process.pid, value) + except ProcessLookupError: + pass + + def close_without_group_authority() -> None: + _close_direct_child(process, deadline=deadline) + if _probe_group_exists(process.pid, deadline=deadline): + raise ContractError("diagnostic process group cleanup unconfirmed") try: + if start_identity is None: + close_without_group_authority() + return if owned_group_exists(): signal_group(signal.SIGTERM) grace = min(deadline, time.monotonic() + PROBE_TERM_GRACE_SECONDS) @@ -105,6 +227,8 @@ def signal_group(value: int) -> None: if remaining <= 0: raise ContractError("diagnostic cleanup deadline expired") process.wait(timeout=remaining) + except _ProbeOwnershipUnavailable: + close_without_group_authority() finally: for stream in (process.stdin, process.stdout, process.stderr): if stream is not None: diff --git a/test/core/test_diagnostics.py b/test/core/test_diagnostics.py index ad9e67c..3121ef8 100644 --- a/test/core/test_diagnostics.py +++ b/test/core/test_diagnostics.py @@ -370,6 +370,354 @@ def test_reused_probe_identity_is_not_safe_to_signal(self): signal_group.assert_not_called() process.stdout.close.assert_called_once() + def test_missing_capture_and_observation_never_authorize_group_signals(self): + process = mock.Mock(pid=987654, returncode=None, stderr=None) + with ( + mock.patch.object(probe_process, "PROBE_CLEANUP_SECONDS", 0.05), + mock.patch.object(probe_process, "PROBE_TERM_GRACE_SECONDS", 0.01), + mock.patch.object(probe_process, "_probe_group_exists", return_value=True), + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg") as signal_group, + self.assertRaises(ContractError), + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + process.wait.assert_not_called() + process.stdout.close.assert_called_once() + + def test_known_capture_missing_observation_has_no_unproven_group_authority(self): + for returncode in (None, 0): + with self.subTest(returncode=returncode): + process = mock.Mock(pid=987654, returncode=returncode, stderr=None) + with ( + mock.patch.object(probe_process, "_probe_group_exists", return_value=True), + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg") as signal_group, + mock.patch.object(probe_process.os, "kill") as signal_child, + self.assertRaises(ContractError), + ): + probe_process.close_probe(process, start_identity="captured-start") + signal_group.assert_not_called() + signal_child.assert_not_called() + process.wait.assert_not_called() + process.stdout.close.assert_called_once() + + def test_probe_does_not_read_output_without_captured_identity(self): + process = mock.Mock(pid=987654, returncode=None, stderr=None) + process.wait.return_value = 0 + with ( + mock.patch.object(diagnostics.subprocess, "Popen", return_value=process), + mock.patch.object(diagnostics, "capture_probe_identity", return_value=None), + mock.patch.object(diagnostics, "_close_probe") as close, + mock.patch.object(diagnostics.selectors, "DefaultSelector") as selector, + mock.patch.object(diagnostics.os, "read", return_value=b"") as read, + self.assertRaisesRegex(ContractError, "ownership is unavailable"), + ): + diagnostics._probe_output(["/fixture/claude", "auth", "status", "--json"], + project=self.project, environment=diagnostics._environment(self.home)) + selector.assert_not_called() + read.assert_not_called() + process.wait.assert_not_called() + close.assert_called_once_with(process, start_identity=None) + + def test_codex_auth_does_not_construct_or_send_protocol_without_identity(self): + process = mock.Mock(pid=987654, returncode=None, stderr=None) + with ( + mock.patch.object(diagnostics.subprocess, "Popen", return_value=process), + mock.patch.object(diagnostics, "capture_probe_identity", return_value=None), + mock.patch.object(diagnostics, "_close_probe") as close, + mock.patch.object(diagnostics, "JsonLinePeer") as peer, + mock.patch.object(diagnostics, "receive_response", side_effect=[ + {"result": {}}, {"result": {"account": {"type": "chatgpt"}, "requiresOpenaiAuth": True}}, + ]) as receive, + ): + authentication = diagnostics._codex_auth("/fixture/codex", project=self.project, + environment=diagnostics._environment(self.home)) + self.assertIsNone(authentication["authenticated"]) + self.assertFalse(authentication["subscription_supported"]) + peer.assert_not_called() + receive.assert_not_called() + close.assert_called_once_with(process, start_identity=None) + + def test_unidentified_live_child_is_reaped_without_group_signals(self): + ready = self.root / "unidentified-ready" + code = ( + "import signal,time\nfrom pathlib import Path\n" + "signal.signal(signal.SIGTERM, signal.SIG_IGN)\n" + f"Path({str(ready)!r}).touch()\n" + "time.sleep(60)\n" + ) + process = diagnostics.subprocess.Popen( + [sys.executable, "-c", code], stdin=diagnostics.subprocess.DEVNULL, + stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, + start_new_session=True, + ) + try: + deadline = time.monotonic() + 2 + while not ready.exists() and time.monotonic() < deadline: + time.sleep(0.01) + self.assertTrue(ready.exists()) + with ( + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg", wraps=os.killpg) as signal_group, + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + self.assertIsNotNone(process.returncode) + self.assertFalse(_live_group_exists(process.pid)) + self.assertTrue(process.stdout.closed) + finally: + if process.poll() is None: + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_missing_kernel_child_authority_denies_all_signals(self): + process = diagnostics.subprocess.Popen( + [sys.executable, "-c", "import time; time.sleep(60)"], + stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, + stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, + ) + try: + with ( + mock.patch.object(probe_process.os, "waitpid", side_effect=ChildProcessError), + mock.patch.object(probe_process.os, "killpg") as signal_group, + mock.patch.object(probe_process.os, "kill") as signal_child, + self.assertRaisesRegex(ContractError, "ownership is unavailable"), + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + signal_child.assert_not_called() + self.assertIsNone(process.returncode) + self.assertTrue(process.stdout.closed) + finally: + process.kill() + process.wait(timeout=2) + process.stdout.close() + self.assertFalse(_live_group_exists(process.pid)) + + def test_known_capture_without_waitid_falls_back_to_verified_child_only(self): + process = diagnostics.subprocess.Popen( + [sys.executable, "-c", "import time; time.sleep(60)"], + stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, + stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, + ) + captured = process_start_identity(process.pid) + self.assertIsNotNone(captured) + try: + with ( + mock.patch.object(probe_process.os, "waitid", None, create=True), + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg", wraps=os.killpg) as signal_group, + ): + probe_process.close_probe(process, start_identity=captured) + signal_group.assert_not_called() + self.assertIsNotNone(process.returncode) + self.assertFalse(_live_group_exists(process.pid)) + self.assertTrue(process.stdout.closed) + finally: + if process.poll() is None: + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_ps_zombie_anchor_rejects_reparented_wrong_group_and_malformed_rows(self): + process = diagnostics.subprocess.Popen( + [sys.executable, "-c", "import time; time.sleep(60)"], + stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, + stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, + ) + pid, parent = process.pid, os.getpid() + cases = [ + (0, f"{pid} {parent + 1} {pid} Z\n"), + (0, f"{pid} {parent} {pid + 1} Z\n"), + (0, f"{pid + 1} {parent} {pid} Z\n"), + (0, f"{pid} {parent} {pid} S\n"), + (0, f"{pid} {parent} {pid}\n"), + (0, f"{pid} {parent} {pid} Z\n{pid} {parent} {pid} Z\n"), + (0, "private-banner-token\n"), + (0, "x" * 1025), + (1, f"{pid} {parent} {pid} Z\n"), + ] + try: + for code, output in cases: + with self.subTest(code=code, output=output[:100]): + with ( + mock.patch.object(probe_process.os, "waitid", None, create=True), + mock.patch.object(probe_process.subprocess, "run", return_value=mock.Mock(returncode=code, stdout=output)) as inventory, + mock.patch.object(probe_process.os, "killpg") as signal_group, + ): + self.assertFalse(probe_process._retained_child_anchor(process, deadline=time.monotonic() + 1)) + signal_group.assert_not_called() + self.assertLessEqual(inventory.call_args.kwargs["timeout"], 0.5) + self.assertEqual(set(inventory.call_args.kwargs["env"]), {"HOME", "USER", "PATH"}) + for error in (OSError, diagnostics.subprocess.TimeoutExpired("/bin/ps", 0.5)): + with self.subTest(error=type(error).__name__): + with ( + mock.patch.object(probe_process.os, "waitid", None, create=True), + mock.patch.object(probe_process.subprocess, "run", side_effect=error), + ): + self.assertFalse(probe_process._retained_child_anchor(process, deadline=time.monotonic() + 1)) + finally: + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_cleanup_reap_lock_contention_cannot_exceed_deadline_or_signal(self): + process = diagnostics.subprocess.Popen( + [sys.executable, "-c", "import time; time.sleep(60)"], + stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, + stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, + ) + process._waitpid_lock.acquire() + started = time.monotonic() + try: + with ( + mock.patch.object(probe_process, "PROBE_CLEANUP_SECONDS", 0.05), + mock.patch.object(probe_process.os, "killpg") as signal_group, + mock.patch.object(probe_process.os, "kill") as signal_child, + self.assertRaisesRegex(ContractError, "deadline expired"), + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + signal_child.assert_not_called() + self.assertLess(time.monotonic() - started, 0.5) + self.assertTrue(process.stdout.closed) + finally: + process._waitpid_lock.release() + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_probe_eof_retains_exited_parent_until_owned_descendant_cleanup(self): + record = self.root / "eof-child.json" + code = ( + "import json,os,signal,time\nfrom pathlib import Path\n" + f"import sys\nsys.path.insert(0, {str(ROOT / 'plugin/core/src')!r})\n" + "from devsquad.supervisor import process_start_identity\n" + "read_ready,write_ready=os.pipe()\n" + "if os.fork()==0:\n" + " os.close(read_ready)\n" + " os.dup2(os.open(os.devnull,os.O_WRONLY),1)\n" + " signal.signal(signal.SIGTERM, signal.SIG_IGN)\n" + f" Path({str(record)!r}).write_text(json.dumps({{'pid':os.getpid(),'start':process_start_identity(os.getpid())}}))\n" + " os.write(write_ready,b'r')\n os.close(write_ready)\n time.sleep(60)\n os._exit(0)\n" + "os.close(write_ready)\nos.read(read_ready,1)\nos.close(read_ready)\n" + "print('codex-cli 0.159.2',flush=True)\nos._exit(0)\n" + ) + captured_processes = [] + real_spawn, real_close = diagnostics.subprocess.Popen, diagnostics._close_probe + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + captured_processes.append(process) + return process + def close(process, **kwargs): + self.assertIsNone(process.returncode, "probe reaped its ownership anchor before cleanup") + real_close(process, **kwargs) + try: + with ( + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + mock.patch.object(diagnostics, "_close_probe", side_effect=close), + ): + code, output = diagnostics._probe_output([sys.executable, "-c", code], + project=self.project, environment=diagnostics._environment(self.home)) + self.assertEqual((code, output.strip()), (0, "codex-cli 0.159.2")) + self.assertFalse(_live_group_exists(captured_processes[0].pid)) + finally: + child = json.loads(record.read_text()) if record.exists() else None + if child and process_start_identity(child["pid"]) == child["start"]: + try: + os.kill(child["pid"], signal.SIGKILL) + except ProcessLookupError: + pass + if captured_processes: + process = captured_processes[0] + if process.poll() is None: + process.kill() + process.wait(timeout=2) + process.stdout.close() + + def test_unidentified_descendant_cleanup_is_unconfirmed_without_group_signals(self): + record = self.root / "unidentified-child.json" + code = ( + "import json,os,signal,time\nfrom pathlib import Path\n" + f"import sys\nsys.path.insert(0, {str(ROOT / 'plugin/core/src')!r})\n" + "from devsquad.supervisor import process_start_identity\n" + "read_ready,write_ready=os.pipe()\n" + "if os.fork()==0:\n" + " os.close(read_ready)\n" + " signal.signal(signal.SIGTERM, signal.SIG_IGN)\n" + f" Path({str(record)!r}).write_text(json.dumps({{'pid':os.getpid(),'start':process_start_identity(os.getpid())}}))\n" + " os.write(write_ready,b'r')\n os.close(write_ready)\n time.sleep(60)\n os._exit(0)\n" + "os.close(write_ready)\nos.read(read_ready,1)\nos.close(read_ready)\ntime.sleep(60)\n" + ) + process = diagnostics.subprocess.Popen( + [sys.executable, "-c", code], stdin=diagnostics.subprocess.DEVNULL, + stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, + start_new_session=True, + ) + try: + deadline = time.monotonic() + 2 + while not record.exists() and time.monotonic() < deadline: + time.sleep(0.01) + self.assertTrue(record.exists()) + with ( + mock.patch.object(probe_process, "process_start_identity", return_value=None), + mock.patch.object(probe_process.os, "killpg", wraps=os.killpg) as signal_group, + self.assertRaisesRegex(ContractError, "cleanup unconfirmed"), + ): + probe_process.close_probe(process, start_identity=None) + signal_group.assert_not_called() + self.assertIsNotNone(process.returncode) + self.assertTrue(_live_group_exists(process.pid)) + self.assertTrue(process.stdout.closed) + finally: + child = json.loads(record.read_text()) if record.exists() else None + if child and process_start_identity(child["pid"]) == child["start"]: + try: + os.kill(child["pid"], signal.SIGKILL) + except ProcessLookupError: + pass + if process.poll() is None: + process.kill() + process.wait(timeout=2) + process.stdout.close() + deadline = time.monotonic() + 2 + while _live_group_exists(process.pid) and time.monotonic() < deadline: + time.sleep(0.02) + self.assertFalse(_live_group_exists(process.pid)) + + def test_doctor_unidentified_auth_is_unknown_and_every_owned_parent_is_absent(self): + self.codex() + self.install("claude", response={"loggedIn": True, "authMethod": "claude.ai"}) + processes = [] + real_spawn, real_capture = diagnostics.subprocess.Popen, diagnostics.capture_probe_identity + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + processes.append(process) + return process + def capture(process): + return real_capture(process) if process.args[-1] == "--version" else None + with ( + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + mock.patch.object(diagnostics, "capture_probe_identity", side_effect=capture), + mock.patch.object(diagnostics, "JsonLinePeer") as peer, + ): + report = self.report() + peer.assert_not_called() + self.assertFalse(report["ready"]) + self.assertFalse(report["adapter_ready"]) + for name in ("codex", "claude"): + row = self.row(report, name) + self.assertTrue(row["supported"]) + self.assertIsNone(row["authenticated"]) + self.assertFalse(row["ready"]) + for process in processes: + self.assertIsNotNone(process.returncode) + self.assertFalse(_live_group_exists(process.pid)) + if process.stdout is not None: + self.assertTrue(process.stdout.closed) + def test_missing_group_is_confirmed_before_reap_without_signaling_stale_pid(self): process = mock.Mock(pid=987654, returncode=0, stderr=None) with ( diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py index 5f7c59f..52c7b99 100644 --- a/test/core/test_task_entry.py +++ b/test/core/test_task_entry.py @@ -856,7 +856,18 @@ def native_popen(argv, *positional, **keywords): from devsquad.task_entry import discover_models as real_discover_models def discover_then_confirm_parent_exit(*arguments, **keywords): models = real_discover_models(*arguments, **keywords) - self.assertEqual(spawned["process"].wait(timeout=2), 0) + # Observe exit without wait/poll: the unreaped child reserves its + # PID while shared cleanup owns the TERM-ignoring descendant. + deadline = time.monotonic() + 2 + observed = "" + while time.monotonic() < deadline: + observed = subprocess.run(["/bin/ps", "-p", str(spawned["process"].pid), "-o", "stat="], + text=True, capture_output=True, timeout=1).stdout.strip() + if observed.startswith("Z"): + break + time.sleep(0.01) + self.assertTrue(observed.startswith("Z"), "provider parent did not exit before cleanup") + self.assertIsNone(spawned["process"].returncode) return models try: with ( From fb64e43a28e9b8eab6670be3d7da8e4fd0fcaa97 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 17:20:57 -0700 Subject: [PATCH 188/197] WIP checkpoint: R6/C1: preserve repaired release candidate, package all schemas and prepare public0.11 core release (2026-10-02 17:20) --- .claude-plugin/marketplace.json | 6 +- CHANGELOG.md | 33 ++ README.md | 52 +- docs/RUNTIME-GUIDE.md | 74 ++- docs/generated/core-reference.md | 9 +- docs/plans/engineering-team/CONTRACTS.md | 44 ++ docs/plans/engineering-team/RESUME.md | 211 +++++++- docs/plans/engineering-team/backlog.json | 2 +- ...terminal-readiness-partial-2026-10-02.json | 30 +- .../evidence/R7-NATIVE-BOUNDARY-BLOCKER.md | 51 ++ .../r7-council-reconciliation-partial.json | 103 ++++ ...inal-nongenerating-network-diagnostic.json | 153 ++++++ ...r7-public-fixture-workflow-comparison.json | 204 ++++++++ ...ublic-fixture-workflow-predeclaration.json | 52 ++ .../r7-temporary-install-epoch-proof.json | 30 ++ plugin/.claude-plugin/plugin.json | 4 +- plugin/core/pyproject.toml | 2 +- plugin/core/schemas/council.schema.json | 17 + plugin/core/schemas/policy.schema.json | 2 +- plugin/core/schemas/task.schema.json | 7 +- plugin/core/src/devsquad/cli.py | 136 +++++- .../core/src/devsquad/codex_review_worker.py | 38 +- plugin/core/src/devsquad/contracts.py | 4 +- plugin/core/src/devsquad/council.py | 242 ++++++++++ .../core/src/devsquad/council_comparison.py | 166 +++++++ plugin/core/src/devsquad/council_isolation.py | 141 ++++++ plugin/core/src/devsquad/council_runtime.py | 452 ++++++++++++++++++ .../core/src/devsquad/council_task_entry.py | 74 +++ plugin/core/src/devsquad/council_worker.py | 97 ++++ plugin/core/src/devsquad/detached.py | 30 +- plugin/core/src/devsquad/diagnostics.py | 16 +- plugin/core/src/devsquad/learning.py | 2 +- plugin/core/src/devsquad/mcp_server.py | 13 +- .../migrations/017_council_contract_epoch.sql | 5 + plugin/core/src/devsquad/probe_process.py | 35 ++ plugin/core/src/devsquad/reports.py | 5 +- plugin/core/src/devsquad/router.py | 17 + plugin/core/src/devsquad/service.py | 144 +++++- plugin/core/src/devsquad/store.py | 117 ++++- plugin/core/src/devsquad/supervisor.py | 30 +- plugin/core/src/devsquad/task_entry.py | 15 +- plugin/core/src/devsquad/validation.py | 25 +- test/core/test_capacity.py | 4 +- test/core/test_council_comparison.py | 92 ++++ test/core/test_council_contract.py | 79 +++ test/core/test_council_entry.py | 84 ++++ test/core/test_council_epoch.py | 58 +++ test/core/test_council_install_epoch.py | 229 +++++++++ test/core/test_council_isolation.py | 105 ++++ test/core/test_council_reconciliation.py | 229 +++++++++ test/core/test_council_runtime.py | 314 ++++++++++++ test/core/test_decision_store.py | 3 +- test/core/test_diagnostics.py | 143 +++++- test/core/test_experiment_assignment_store.py | 4 +- test/core/test_handoff_store.py | 4 +- test/core/test_learning.py | 4 +- test/core/test_lifecycle.py | 3 +- test/core/test_mcp.py | 13 + test/core/test_task_entry.py | 4 +- 59 files changed, 4144 insertions(+), 118 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/R7-NATIVE-BOUNDARY-BLOCKER.md create mode 100644 docs/plans/engineering-team/evidence/r7-council-reconciliation-partial.json create mode 100644 docs/plans/engineering-team/evidence/r7-final-nongenerating-network-diagnostic.json create mode 100644 docs/plans/engineering-team/evidence/r7-public-fixture-workflow-comparison.json create mode 100644 docs/plans/engineering-team/evidence/r7-public-fixture-workflow-predeclaration.json create mode 100644 docs/plans/engineering-team/evidence/r7-temporary-install-epoch-proof.json create mode 100644 plugin/core/schemas/council.schema.json create mode 100644 plugin/core/src/devsquad/council.py create mode 100644 plugin/core/src/devsquad/council_comparison.py create mode 100644 plugin/core/src/devsquad/council_isolation.py create mode 100644 plugin/core/src/devsquad/council_runtime.py create mode 100644 plugin/core/src/devsquad/council_task_entry.py create mode 100644 plugin/core/src/devsquad/council_worker.py create mode 100644 plugin/core/src/devsquad/migrations/017_council_contract_epoch.sql create mode 100644 test/core/test_council_comparison.py create mode 100644 test/core/test_council_contract.py create mode 100644 test/core/test_council_entry.py create mode 100644 test/core/test_council_epoch.py create mode 100644 test/core/test_council_install_epoch.py create mode 100644 test/core/test_council_isolation.py create mode 100644 test/core/test_council_reconciliation.py create mode 100644 test/core/test_council_runtime.py diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 83ce1a8..4021e6b 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -5,14 +5,14 @@ }, "metadata": { "description": "DevSquad - Engineering Manager that coordinates AI coding agents", - "version": "0.10.0" + "version": "0.11.0" }, "plugins": [ { "name": "devsquad", "source": "./plugin", "description": "Engineering Manager that coordinates AI coding agents through enforced delegation", - "version": "0.10.0" + "version": "0.11.0" } ] -} \ No newline at end of file +} diff --git a/CHANGELOG.md b/CHANGELOG.md index 3e682fe..51d7da0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,39 @@ All notable changes to DevSquad are documented here. > **Note:** Project was renumbered from 2.x to 0.x semver in Feb 2026 to reflect pre-stable status. Entries below have been renumbered accordingly. +## [0.11.0] — 2026-10-02 + +### Added +- Standalone dependency-free Python core (`squad` 0.1.0) with durable saved + runs, isolated candidates, bounded Claude implementation, independent Codex + review, candidate-bound checks and fenced host acceptance. +- Guided setup, readiness, review/fix, status/result, finish, cancel and safe + resume commands; optional pinned MCP transport for the same local ledger. +- Catalog-backed profiles, shared subscription-pool capacity fences, explicit + outcome/trial reports and guarded qualification/promotion/rollback. +- Reversible content-addressed installation and guarded schema-17 upgrade. + The wheel packages every runtime schema, including the experimental Council + contract. Old schema-16 clients cannot mutate the upgraded ledger. + +### Fixed +- Interrupted guided completion now saves exact intent and recovers only + matching authority, retaining prior expiry rejection history. +- Native probes fail closed without process identity. Bounded natural exit + preserves version/auth exit status before owned cleanup; controlled Python + fixtures cannot invalidate a pristine candidate with undeclared bytecode. +- Bash 3.2 compatibility, optional-jq behavior and legacy wrapper error + contracts remain supported. + +### Limitations +- Council is experimental and native-unavailable; automatic invocation is OFF. + Its fixture comparison is inconclusive, not a quality or savings claim. +- Jev routing remains OFF. Provider operations depend on installed harness + capabilities and subscription authentication; paid API fallback is not used. +- Claude CLI handoff and Claude-to-Codex delivery have live receipts. Claude + desktop Code-tab proof is separate and unverified. Antigravity means the + `agy` CLI, not its IDE; Grok/Antigravity MCP status does not prove automatic + implementation/review roles. See the runtime guide for supported boundaries. + ## [0.10.0] — 2026-07-07 ### Maintainability & scalability review diff --git a/README.md b/README.md index ca824d4..c2b42b4 100644 --- a/README.md +++ b/README.md @@ -2,11 +2,13 @@ # DevSquad -### Your AI coding agent ignores your rules. Hooks don't. +### A local engineering team with saved work and verifiable results -**DevSquad turns Claude Code into an engineering manager that _physically intercepts_ tool calls and routes the grunt work to Gemini, Codex, and Grok — then runs a live A/B test on whether that even helps.** +**DevSquad coordinates bounded implementation, independent review, tests and +host acceptance across your installed AI harnesses. Runs and evidence survive +closing a client. The standalone runtime does not require Claude's plugin.** -[![tests](https://img.shields.io/badge/tests-177%20passing-brightgreen)](test/) +[![tests](https://github.com/joshidikshant/devsquad/actions/workflows/offline.yml/badge.svg)](https://github.com/joshidikshant/devsquad/actions/workflows/offline.yml) [![bash](https://img.shields.io/badge/bash-3.2%2B-blue)](CONTRIBUTING.md) [![jq](https://img.shields.io/badge/jq-optional-blue)](CONTRIBUTING.md) [![license](https://img.shields.io/badge/license-MIT-black)](LICENSE) @@ -22,12 +24,48 @@ --- -**Engineering-team build in progress:** continuing after an interrupted AI -session? Read the [recovery checkpoint](docs/plans/engineering-team/RESUME.md) -for saved work, verified results and the next action. The product described -below is the existing plugin; the new runner's status is tracked in the +## Standalone runtime quickstart + +Requires a Unix-like host, Python 3.11+ and existing provider CLI subscription +logins. Installation is local and does not download Python dependencies: + +```bash +git clone https://github.com/joshidikshant/devsquad.git +cd devsquad +./install.sh --core-only +export PATH="$HOME/.local/bin:$PATH" +squad doctor +squad review --base main --dry-run +``` + +In the project you want to work on, commit the inputs, inspect the dry run, +then use `squad review --base main --wait` or +`squad fix "the bounded issue" --write-path src --wait`. The saved run stops +at a host handoff with checks and the exact finish command; assess the evidence +before accepting. `squad status`, `squad result RUN_ID`, `squad cancel RUN_ID` +and `squad resume RUN_ID` operate on the same saved work. + +For Codex/Claude/Antigravity CLI/Grok MCP setup, safe updates, supported +operations and troubleshooting, read the [runtime guide](docs/RUNTIME-GUIDE.md). +The optional MCP transport needs an explicitly prepared pinned wheelhouse; +it is not silently downloaded. Existing paid accounts do not imply that every +harness supports every worker role. No paid API fallback is enabled. + +Council is native-unavailable and automatic use is OFF. Jev routing is OFF; +no production quality, cost savings or Plus-window savings are claimed. +Claude CLI handoff and implementation → independent Codex review → tests have +live receipts; Claude desktop Code-tab proof remains unverified. Antigravity +means `agy`, not the IDE. + +Continuing development after an interruption? Read the +[recovery checkpoint](docs/plans/engineering-team/RESUME.md) and [implementation plan](docs/plans/engineering-team/START-HERE.md). +## Legacy Claude plugin + +The sections below describe the separately supported Claude hook plugin, +not automatic worker-role coverage in the standalone runtime. + ## The 30-second version You told Claude to delegate the boring stuff. It nodded. Then it read 40 files itself, blew through its context window, and you paid for every token. diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index 5937dd6..6bf3a83 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -202,6 +202,66 @@ Every surface uses these operations directly or through the thin local MCP bridge described in [MCP-LOCAL-ACCESS.md](plans/engineering-team/MCP-LOCAL-ACCESS.md). Closing an app does not cancel the detached run. +## Manual Council (partial source implementation) + +Council reuses the saved runner with two independent read-only proposers, a +distinct critic and the existing sole lead. Automatic triggering is always off; +there is one round and no internal retry. Prepare a scoped question without task +JSON: + +```bash +squad council "Which retry policy avoids duplicate side effects?" \ + --read-path src/retry.py --lead host --max-invocations 3 --dry-run +``` + +Inspect the exact commits, read scope, rubric, checks and three catalog model +identities. Repeat `--model MODEL_ID` exactly three times to pin `proposer_a`, +`proposer_b` and critic; catalog availability is not tested quality qualification. +`--criterion ID=DESCRIPTION`, `--evidence ARTIFACT_ID:SHA256` and `--check COMMAND` +freeze explicit rubric, saved evidence and bounded checks. Preparation performs +no generation. Remove `--dry-run` only after doctor and per-run capability gates +are satisfied. + +Native Council currently fails closed before any role generation: +`native_ready:false` means a genuine nongenerating backend response under the +exact default-deny macOS sandbox has not been attested. Bootstrap or cached +catalog success is not that proof. Unsupported operating systems have no unsafe +fallback. Normal review/fix remain separate. Do not broaden filesystem/MCP +access, bypass provider permissions or use a paid API to work around this gate. + +Once the gated workflow is available, a host-led run pauses with shuffled A/B +proposals, criterion assessments, mandatory checks and retained dissent. Status +does not choose a proposal or infer a disposition. Finish explicitly: + +```bash +squad status RUN_ID +squad finish RUN_ID --accept --choose synthesis \ + --reason "Retain the idempotency and expiry safeguards." \ + --supported-claim "Never retry a non-idempotent request blindly." \ + --discarded-alternative "Blind retry without a stable key." \ + --validation "Exercise duplicate requests and key-retention expiry." +squad result RUN_ID +``` + +Choose `A`, `B` or `synthesis`; repeat supported/discarded flags as needed. Every +critic objection remains in the decision. Reject with explicit assessment when +appropriate; votes cannot override failed checks. Extra deliberation requires a +new capped run, not `--revise`. `council-finish` is an explicit-choice alias; +omitted IDs follow the same unique-current-project rule as normal finish. +An interrupted guided finish exposes an exact retry command, including every +choice input. Only the latest matching canonical guided claim can recover an +expired fence; an app-owned claim, even named `terminal-operator`, cannot be +adopted. A submitted decision continues with `squad resume RUN_ID`. + +The default headless lead uses four capped worker invocations, chooses only from +validated saved evidence and continues its own handoff with `--wait` or resume; +a host cannot claim it. Missing participants, invalid identities, quota, +cancellation and check failures remain failures, never fabricated consensus. +The retained predeclared matched/held-out public fixture comparison is +inconclusive and proves mechanics only: native quality, escaped defects, rework +and native allowance remain unknown. It authorizes no automatic use or model +promotion. + ## Recover or cancel Never delete the runtime database, an active release or a run-owned worktree @@ -234,8 +294,13 @@ install a new source digest and retain the old directory for run evidence. ## Supported and deferred boundaries -The current packaged contract is Python 3.11+, public JSON contract version 1, -SQLite schema 16 and optional MCP SDK 2.2.0 exactly. Native Codex fixtures and +The source candidate contract is Python 3.11+, public JSON contract version 1, +SQLite schema 17 and optional MCP SDK 2.2.0 exactly. The accepted installed R5 +boundary remains schema 16 until an explicitly accepted Council update; R6 +source also uses epoch 16, with its own acceptance tracked separately. This +guide does not claim that source Council is installed or live accepted. Epoch 17 +fences old clients that cannot understand Council handoff authority, and its +installer defers on active or recoverable schema-16 work. Native Codex fixtures and recorded live proofs cover bundled `codex-cli 0.153.4` and `codex-cli 0.155.0-alpha.9.2`; the latter passed a fresh native initialize and complete model-catalog probe. The resolver prefers that verified bundled @@ -280,5 +345,6 @@ transfer. The one authorized Jev pilot is recorded separately and runtime classification remains off; normal routing uses the deterministic zero-call path. Laya is not -installed unless its declared trigger fires. C1 Council implementation and -acceptance remain pending in the full engineering-team delivery. +installed unless its declared trigger fires. C1 Council is implemented partially +in source with offline process/isolation/epoch proofs; native backend attestation, +live quality and final installed acceptance remain open. diff --git a/docs/generated/core-reference.md b/docs/generated/core-reference.md index 84e1483..a53d714 100644 --- a/docs/generated/core-reference.md +++ b/docs/generated/core-reference.md @@ -10,9 +10,11 @@ Regenerate with `python3 scripts/generate-core-reference.py`; verify with - `squad capacity [-h] {observe} ...` - `squad capacity observe [-h] --file FILE [--json] [--runtime-dir RUNTIME_DIR]` - `squad classify [-h] [--cwd CWD] [--model MODEL] [--effort EFFORT] [--permission {read_only,workspace_write}] [--timeout TIMEOUT] [--transport {cli_exec,native_protocol}] [--catalog-file CATALOG_FILE] --returncode RETURNCODE --stdout-file STDOUT_FILE --stderr-file STDERR_FILE {codex,antigravity,grok}` +- `squad council [-h] [--project-dir PROJECT_DIR] [--base BASE] [--target TARGET] [--read-path READ_PATH] [--evidence EVIDENCE] [--criterion CRITERION] [--check CHECK] [--model MODEL] [--effort EFFORT] [--lead {headless,host}] [--max-invocations MAX_INVOCATIONS] [--idempotency-key IDEMPOTENCY_KEY] [--dry-run] [--wait] [--json] [--runtime-dir RUNTIME_DIR] question` +- `squad council-finish [-h] (--accept | --reject) --reason REASON --choose {A,B,synthesis} [--supported-claim SUPPORTED_CLAIM] [--discarded-alternative DISCARDED_ALTERNATIVE] --validation VALIDATION [--json] [--project-dir PROJECT_DIR] [--runtime-dir RUNTIME_DIR] [run]` - `squad doctor [-h] [--json] [--project-dir PROJECT_DIR] [--squad-executable SQUAD_EXECUTABLE]` - `squad events [-h] [--after AFTER] [--limit LIMIT] [--json] [--runtime-dir RUNTIME_DIR] run` -- `squad finish [-h] (--accept | --reject | --revise) --reason REASON [--project-dir PROJECT_DIR] [--json] [--runtime-dir RUNTIME_DIR] [run]` +- `squad finish [-h] (--accept | --reject | --revise) --reason REASON [--choose {A,B,synthesis}] [--supported-claim SUPPORTED_CLAIM] [--discarded-alternative DISCARDED_ALTERNATIVE] [--validation VALIDATION] [--project-dir PROJECT_DIR] [--json] [--runtime-dir RUNTIME_DIR] [run]` - `squad fix [-h] [--base BASE] [--target TARGET] [--project-dir PROJECT_DIR] [--write-path WRITE_PATH] [--check CHECK] [--check-timeout CHECK_TIMEOUT] [--review-model REVIEW_MODEL] [--review-effort REVIEW_EFFORT] [--review-mode {standard,adversarial}] [--review-focus REVIEW_FOCUS] [--implementer-model IMPLEMENTER_MODEL] [--implementer-effort IMPLEMENTER_EFFORT] [--idempotency-key IDEMPOTENCY_KEY] [--dry-run] [--wait] [--json] [--runtime-dir RUNTIME_DIR] issue` - `squad handoff [-h] {claim,complete} ...` - `squad handoff claim [-h] --expected-version EXPECTED_VERSION --owner OWNER [--claim-file CLAIM_FILE] [--json] [--runtime-dir RUNTIME_DIR] run` @@ -65,14 +67,15 @@ runtime never infers them from chat history. |---|---|---|---| | `adapter.schema.json` | `https://devsquad.local/schemas/adapter-v1.json` | `schema_version`, `name`, `transport`, `binary_candidates`, `capabilities`, `permission_profiles` | `fc6b811b59bd2920d3dc1588d5bf447515a25d87219d07d76e1764efd60ad9c0` | | `check-result.schema.json` | `https://devsquad.local/schemas/check-result-v2.json` | `schema_version`, `candidate_sha256`, `target_oid`, `id`, `argv`, `cwd`, `required_to_pass`, `status`, `returncode`, `error_code`, `duration_ms`, `stdout`, `stderr` | `5145755d8e60503151f8938f1753c657724f1a2de3a442500c14eda158fe04a1` | +| `council.schema.json` | `https://devsquad.local/schemas/council-v1.json` | `schema_version`, `enabled`, `automatic`, `reason`, `min_valid_proposals`, `required_critics`, `max_invocations`, `seed`, `evidence`, `rubric` | `8d72642179ce163a707168ad5e63e3e0cff794ce32afd422f232bfa5af13dff5` | | `execution-identity.schema.json` | `https://devsquad.local/schemas/execution-identity-v1.json` | `harness`, `harness_version`, `model_provider`, `model_family`, `model`, `effort`, `tools`, `permissions`, `account_pool`, `verification` | `99386fb81c3567bdd3c880b37d363d5ca61cccb3ff35add5eb01a1ad916dfa90` | | `launch-spec.schema.json` | `https://devsquad.local/schemas/launch-spec-v1.json` | `schema_version`, `adapter`, `transport`, `argv`, `cwd`, `stdin_path`, `timeout_seconds`, `requested`, `environment` | `12e2c3ffbfee6b3a9e93e41d9dd7d6a2c82a9d62e981d9378307b12f6ff12f5c` | | `normalized-result.schema.json` | `https://devsquad.local/schemas/normalized-result-v1.json` | `schema_version`, `execution_status`, `error_code`, `output`, `artifact_status`, `acceptance_status`, `requested`, `observed`, `native_ids`, `events` | `8fb7eec0d3cfc952b9fa839b5fdbe950d5cfe06ed69bbbd3bcf835aad9df753d` | -| `policy.schema.json` | `https://devsquad.local/schemas/policy-v1.json` | `schema_version`, `id`, `version`, `roles`, `task_classes`, `require_different_model_for_review`, `account_pools`, `experiment_budget` | `00009a5a1fdfe31b766561e4e890ba75fed25b9692f9732ebf676e63860da50c` | +| `policy.schema.json` | `https://devsquad.local/schemas/policy-v1.json` | `schema_version`, `id`, `version`, `roles`, `task_classes`, `require_different_model_for_review`, `account_pools`, `experiment_budget` | `b706f2fe91b0f22fe4e5d1c6d96ee48a8d5562040c8810c0f100ee9612b725a6` | | `profile.schema.json` | `https://devsquad.local/schemas/profile-v1.json` | `id`, `harness`, `model_family`, `model_id`, `effort`, `required_tools`, `permission_policy`, `account_pool_id`, `billing_mode`, `quality_status`, `evidence_refs` | `fc51b49cd7dd6a134eedf2c2c941301392aef29a2163ac2d307b4fb78509ec96` | | `profiles.schema.json` | `https://devsquad.local/schemas/profiles-v1.json` | `schema_version`, `profiles`, `bindings` | `c14823c89d1b81ee93b74f5bbc2701ae975825db795b5f0937c28fc057d84cf1` | | `review-result.schema.json` | `https://devsquad.local/schemas/review-result-v1.json` | `schema_version`, `candidate_sha256`, `base_oid`, `target_oid`, `review_mode`, `verdict`, `summary`, `findings` | `96b852543f92dca21dafd6c0bc954f8f56d3dbfb0981d601cc05b9c1c9ba6e6b` | -| `task.schema.json` | `https://devsquad.local/schemas/task-v1.json` | `schema_version`, `project`, `workflow`, `goal`, `task_class`, `acceptance`, `checks`, `scope`, `lead`, `routing`, `budget`, `origin` | `afe02492bccc098c444be6095150b9f9af97d1dbda6187c69bff6b8bd0c4eaac` | +| `task.schema.json` | `https://devsquad.local/schemas/task-v1.json` | `schema_version`, `project`, `workflow`, `goal`, `task_class`, `acceptance`, `checks`, `scope`, `lead`, `routing`, `budget`, `origin` | `37537e7f9c21ec07cb9b6a1076a69807278ec682103527059ab0415087ba3da3` | ## Task-shape example diff --git a/docs/plans/engineering-team/CONTRACTS.md b/docs/plans/engineering-team/CONTRACTS.md index 1f80a08..ef2fa39 100644 --- a/docs/plans/engineering-team/CONTRACTS.md +++ b/docs/plans/engineering-team/CONTRACTS.md @@ -368,3 +368,47 @@ pairs and both exact tested fingerprints. A proven bootstrap predecessor without existing explicit baseline contract. Installation of this schema remains gated on old active/recoverable-run upgrade safety; adding the migration does not establish that installation gate. Implementation status is in RESUME.md. + +## Bounded manual Council extension (R7 source gate) + +`council-decision` is an explicitly enabled, automatic-OFF, read-only workflow. +Its strict `CouncilSpec` freezes a bounded question, evidence IDs/hashes, +rubric, seeded labels, two valid proposers, one distinct verified critic and a +worker-invocation cap. Three exact distinct entitled model IDs under the same +verified Codex subscription harness suffice; cross-family preference cannot +substitute for verified entitlement/identity. One round is supported, with no +internal revision or silent retry beyond the frozen fallback budget. + +Each proposer sees only its frozen common evidence. The trusted coordinator +commits each immutable proposal before the critic starts; the critic receives +sanitized seeded A/B proposals without raw identity provenance. The reversible +mapping, actual identities, native IDs, nullable usage, checks and artifacts +are retained separately for audit. Partial anonymity is not a quality claim. +The sole existing lead chooses A, B or synthesis, identifies supported claims, +discarded alternatives, all unresolved objections and validation. Missing, +empty, invalid, cancelled, quota-exhausted or failed roles cannot become quorum; +agreement cannot override failed mandatory checks or integrity constraints. + +Native participant processes require the frozen macOS default-deny Seatbelt +profile and actual per-run own-read/peer-runtime-denial probes. Native Codex +permissions also remain read-only. Saved-artifact MCP defense applies only to +the explicit Council worker context, preserving ordinary worker read behavior; +the OS boundary remains authoritative even without that context marker. +Unsupported OS/capabilities fail unavailable, never launch without isolation. +Bootstrap success alone is not subscription/network or generating readiness. + +Stages reuse existing versioned run/attempt/account ownership and artifact +tables. Submitted inputs have a separate immutable origin digest; stage +projections and imported artifacts are checked under existing transaction +fences. Claim/submission recovery retains exact saved evidence and decision; +guided host intent belongs to the actual acquired/taken-over claim event, not +an owner label or independent intent artifact. Terminal Council reports project +exactly one R5 final outcome with all contributions and truthful missingness. + +Council workflow comparison is separate from R3 profile-binding eligibility: +predeclare matched and held-out workflow cases, preserve input/profile/prompt/ +evidence versions and actual receipt references, and report benefit, harm or +inconclusive without invented scores. Fixture mechanics cannot establish native +quality; automatic triggering remains OFF until a separately accepted gate. +Installed cross-version contract epoch and native/comparison acceptance remain +open at this source checkpoint; see RESUME.md. diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 3a4fc28..4cdcd21 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,43 @@ remain recoverable in Git; detailed receipts and failed gates stay in evidence. ## Current checkpoint — October 2, 2026 +**Public release authorized:** the user selected a public GitHub release in +the existing `joshidikshant/devsquad` repository (no registry or hosted service). +Release version0.11.0 includes core0.1.0; Council stays native-unavailable and +automatic OFF. Whole-plan completion is not claimed. Root owns publication, +final frozen/full/native gates and local install. No tag/release/push yet. + +EOF/fixture repair `4f01456189b38252a678938e7197ac2f12a6838e` is now integrated +at source level: wait for natural exit without reaping under the original +deadline, then owned cleanup; true version/auth status survives delayed exit. +Hung EOF remains bounded/unknown. Controlled Python fixtures use `-B`; native +provider environment and integrity gates are unchanged. Agent strict40 tests +pass on Python3.12/12.894s and3.14/13.398s; root34 diagnostics/11.159s and +eight reconciliation/SDK checks/21.651s pass. Wheel completeness regression +failed as expected (one/2.657s: Council schema missing); packaging fix passes +the same actual installed-wheel test (one/2.831s), no SDK required. + +Final nongenerating Council diagnostic is exhausted: eight denial controls +pass; sandbox backend RPC fails network_request_failed/noHTTP, unsandboxed +control fails strict response validation. Zero generating calls, auth copy +removed and original unchanged. No more network attempts or permission widening. +[Portable evidence](evidence/r7-final-nongenerating-network-diagnostic.json). +Next: checkpoint the coherent integration after Bash/reference gates, prove +the exact trusted check in a pristine candidate worktree, then one narrow EOF +native review and one final frozen combined full suite. Safely update the +local installation and recheck agy CLI; publish only the tested candidate. + +Council reconciliation `0e7dd319b56802329187007738970e26b5609b1d` is integrated, +not accepted/installed. It includes preserved42979ba/72ae413/eee6db7, schema17, +R6 canonical terminal authority, explicit choice/copyable recovery, headless +crash/expiry recovery and shared bootstrap cleanup. Its168 affected tests/ +148.693s (two SDK skips), two actual SDK/2.964s, Bash/reference/diff pass. +Original failed169 loader command and test-only inactive-claim errors are saved +in [reconciliation evidence](evidence/r7-council-reconciliation-partial.json). +Native network remains unavailable before generation; automatic off/comparison +inconclusive. Ownership093 and EOF4f are integrated as described above. +Root owns final frozen full/native/install gates; production remains R5/schema16. + Accepted work must not be restarted or re-tested unchanged: - R3 `39b95f1`: exact native review clean, focused/public/historical evidence @@ -84,8 +121,7 @@ closed. EOF preserves the child PID through cleanup. No ABI additions; helper requires exclusive Popen reaping, never an external waitpid/SIGCHLD reaper. Six red baseline regressions/1.216s; accepted Python3.12 strict36/11.250s and Python3.14 strict36/11.364s pass; root seven/1.259s with exact matching blobs, -Bash/reference/diff green. No Council network -probe ran. Diff/Bash +Bash/reference/diff green. Diff/Bash checks passed, but the required66-test command (89.977s/OK) was **invalidated** by undeclared bytecode; reference was not run. The private invocation now sets PYTHONDONTWRITEBYTECODE=1; candidate integrity was not relaxed. Exact packet @@ -95,9 +131,22 @@ The one frozen baseline full gate completed successfully, exec86382: **554 tests/645.034s**, two optional SDK skips, zero failures/errors/unraisable; UTC/monotonic645.167s agree. RootHEAD32b94d6 source/tests/scripts are equivalent to7130ee7. This is a baseline pass, not proof of the later ownership repair. -No full suite/native run remains active. Next save this ownership integration -after root affected tests/Bash, then run one narrow exact follow-up audit and -the prepared bounded nongenerating resolver-socket diagnostic. Install only after +Follow-up `d2355cd0-603d-4e6a-b1a2-b05b0e99d53a` reviewed exact root8242eff, +**rejected**, failed/version22: medium `R6-EOF-exit-status`. Closing stdout is +not natural exit; immediate cleanup TERM can turn valid version/authentication +into unknown. Readiness repairs bounded nonreaping completion before cleanup. +Its56 tests/27.584s ran OK, but integrity again invalidated source bytecode: +controlled provider fixture children intentionally drop Python env flags and +import the core. Fix those fixture children, not native environment or integrity. +Before another native audit, run the exact trusted check in a pristine offline +candidate worktree and verify the full input fingerprint is unchanged. +Resolver-socket-only nongenerating diagnostic failed actual rate-limits RPC +(-32603/network_request_failed, no HTTP response); all eight access/child probes +and owned cleanup passed, temporary auth copy removed. No generating request. +Council agent owns one paired bounded narrow-DNS diagnostic, never broad access. +No full/native run remains active. Next save integrated Council after root +affected/Bash gate, integrate the EOF/fixture repair, then pristine check proof, +narrow exact native follow-up and one final frozen combined full gate. Install only after final frozen/native acceptance. Production remains accepted R5/schema16. R7 Council remains in isolated `r7-council`; its controlled stage flow is partial. Its source is checkpointed at `42979ba4` with 21 Council tests and @@ -123,10 +172,9 @@ without selector/launcher/schema changes; safe cancel/reconcile permits17, then immutable long-lived/fresh old16 clients deny Council mutations. Two focused tests/7.976s, Bash/reference/diff pass; no production update occurred. Readiness independently verified old16 denial on17 and the actual fourth lead -succeeding. Its one bounded Unix mDNS resolver-socket diagnostic is paused for -the P1 ownership repair. Council agent now owns shared-hunk/formatter and R6 -authority reconciliation on isolated `codex/council-integration` in the former -r6-terminal worktree; root remains on codex/engineering-team. No native Council +succeeding. Shared-hunk/formatter/R6 authority reconciliation0e7dd3 is integrated +as described above; original isolated branches are retained. Root remains on +codex/engineering-team. No native Council generation occurred. Preserve all original checkpoints/attached worktrees. Desktop control worked for scoped inspection. Claude's local Code tab selected only this @@ -251,8 +299,11 @@ was spent; 8/8 family and skill labels, 6/8 tier labels including a confident wrong tier. Runtime classification remains OFF; no adoption benefit or Laya setup is proven. Do not repeat the pilot or silently use hosted/API fallback. -No credit purchases/resets, paid API fallback, global AI settings, push/merge, -deploy or external messages are authorized. Raw provider output and credentials +No credit purchases/resets, paid API fallback, global AI settings or unrelated +external messages are authorized. The user's latest GitHub-release choice +authorizes scoped push/PR/merge/tag/release in joshidikshant/devsquad, superseding +the earlier publication prohibition for that destination only. +Raw provider output and credentials stay outside Git. Native reported cost/usage is not an inspected subscription invoice or a reliable number of Plus five-hour windows. @@ -265,3 +316,141 @@ Private bounded helpers are under observe-run.py, inspect-handoff.py, complete-handoff.py, start-combined-review.py, g4-candidate-cases.py and probe.py. Claims/raw logs remain private; never reuse the completed b950 handoff's prior claim for a different run. + +## Isolated R7 source checkpoint — October 2, 2026 + +The Council slice is prepared in the delegated `r7-council` worktree based on +`bf3de086`, not installed or accepted as native production functionality. It +reuses existing detached ownership, attempt/account reservations, immutable +artifacts, handoff fences and terminal reporting: proposer A, proposer B, +critic, then the existing sole host/headless lead. No daemon, writer or general +scheduler was added. The submitted snapshot remains frozen separately from +fenced stage projections. Automatic triggering is always OFF; one round is +supported (`max_revisions=0`). Another round needs a new explicitly capped run. + +The normal surface is `squad council QUESTION --read-path PATH --max-invocations +4 --dry-run`; three distinct entitled Codex model IDs may share the subscription +harness. Host completion has `council-finish` with an explicit chosen label or +synthesis, claims, discarded alternatives and validation; no manual JSON is +required. Root must reconcile these narrow CLI/service/store hunks with R6's +normal formatter and atomic guided-finish claim marker before integration. + +Actual same-user macOS Seatbelt probes demonstrate default-deny own-evidence +reads while peer evidence, ledger, private logs, artifacts, symlink escapes and +child processes remain denied. Unsupported OS has no unsafe fallback. Native +Codex isolated initialization/catalog listing was non-generating only. Narrow +CFPreferences service/shared-memory support works; HTTPS model refresh still +fails under diagnostic DNS additions. Native subscription/network readiness +and generating receipts are **not proven**; no generating calls were made. +Fixture receipts explicitly retain `all_fixture`, nullable identity/usage and +mechanics-only limitations. The predeclared matched/held-out comparison report, +reserved-launch cancellation probe, cross-version contract-epoch decision, +accepted install/native smoke and integrated full gate are still open. R7/C1 +must remain incomplete in backlog until these gates are independently closed. + +Focused offline tests cover public four-process success, strict quorum/labels, +missing/empty roles, failed mandatory checks, distinct/native identity shape, +objective outcome uniqueness, cancellation, quota exhaustion, restart, guided +claim/submission crashes, expired exact-intent recovery and competing claims. +The final checkpoint message records the exact focused/Bash gate results; do +not infer a full-suite or native-quality pass from this partial source savepoint. + +Checkpoint gates: Council 21 tests passed in 33.055s with ResourceWarnings as +errors; router/validation/task-entry/handoff-service/MCP 67 tests passed in +20.103s (two optional SDK skips). Bash passed all 11 test files; generated +reference and `git diff --check` passed. No whole core suite was run in this +isolated slice. Preserve root's newer R5/R6 evidence when reconciling this note. + +### R7 residual source gates — October 2, 2026 + +The immutable installed R5/schema16 Service reproduced a headless Council claim +that the new Service denies: old owner acquired v28→29; the real fourth/lead +worker completed, but v38 remained awaiting that old owner. New resume correctly +refused to steal the claim. Controlled cancellation reached v44 and all owned +PIDs were absent. This proves a schema16 code guard alone is insufficient. +`017_council_contract_epoch.sql` now advances the contract epoch to17 without +new tables, activating existing exclusive-upgrade deferral and old-connection +write triggers. Historical13/16 assertions remain historical; current-version +assertions use `SUPPORTED_SCHEMA_VERSION`. Source tests prove16 active and +recoverable deferral, terminal reconciliation,17 migration, and old16 fresh/ +preopened writer rejection. Immutable installed16 versus17 confirmation and +installer/package reconciliation remain root-owned acceptance gates. + +Owned gated-launch cancellation is now tested: the actual same-user launcher +is live, cancellation remains nonterminal with no receipt, gate cleanup is +confirmed, then the cancelled Council receipt/final outcome is published. The +unstarted-runner recovery path retains Council reporting without changing the +ordinary M2 ownership boundary or declaring cancellation before cleanup. + +The separate predeclaration/comparison module now executes actual public +branch-review control (one worker) and host Council (three workers) on distinct +matched and held-out questions. It freezes task/profile/policy/prompt versions, +checks original inputs and exact terminal receipt hashes, rejects reused runs +or relabelled identical contracts as held-out, and preserves terminal ledger +immutability. Report conclusion is **inconclusive**, automatic OFF. Accepted +quality, escaped defects, rework, native quota and host usage are unmeasured; +fixture host acceptance is not native quality. Portable declaration/report: +`evidence/r7-public-fixture-workflow-{predeclaration,comparison}.json`. +Canonical private report SHA256: +`ac1deb4ea74ea70b0c337d43d0b0905bb8ed2432445e22cd3c71fb24b5fe8565`. +Exact private receipts/ledger remain at +`/Users/Dikshant/.devsquad/private-probes/devsquad-council-kpfg2i2w`. + +Native Council now fails capability-unavailable **before role launch** while +exact sandbox HTTPS attestation is unverified. Doctor shows implemented partial +mechanics, native-ready false. The last nongenerating probe and demonstrated +versus hypothetical denied facilities are recorded in +`evidence/R7-NATIVE-BOUNDARY-BLOCKER.md`; initialization and fallback/cached +catalog entries cannot lift this gate. Root owns subsequent bounded backend +attestation and MCP inventory/native/install proof; no generating retry or +broad filesystem/network relaxation is authorized by these source results. + +Residual checkpoint gates:25 Council tests passed in90.724s with ResourceWarnings +as errors;35 migration/capacity/decision/lifecycle/learning/assignment tests passed +in54.469s;12 handoff-store tests passed in4.528s;10 supervisor/reservation-crash +tests passed in1.819s. A separate retained public comparison run passed all +assertions and saved the portable report above. Bash all11 files, generated +reference and diff checks passed. Earlier comparison attempts exposed a frozen +package-path mismatch and an attempted write to a terminal run; repaired by +checking the actual packaged module and saving a separate immutable comparison +artifact, not weakening terminal fences. No whole suite/native-quality acceptance +was claimed. Root must reconcile R6 atomic finish markers/human formatter and +complete the install/native/full gates before closing R7/C1. + +### R7 actual temporary installation gate + +The focused installer regression uses `git archive bf3de086:plugin/core`, whose +full source digest exactly matches accepted immutable R5 +`90f1e87fb9acf7756dad46780672de9b84fe75e3631322857a310f089280345e`; +the actual temporary installed old package re-observed code-package digest +`e03bf3a2e362b3aff6d0a5dd00bd837f15afb216d4c58ac1d9c776c4d8f09c67` +and supported/current schema16. No lower-version rewriting was used in this +gate. Temporary MCP installation was intentionally omitted: core/Service/ +database fencing is exercised without network/dependency installation. + +Actual old active and recoverable runs independently deferred the17 candidate +while current selector, launcher, install-state and schema16 stayed unchanged. +Old active work was cancelled with cleanup; recoverable work resumed and +succeeded under the old release. Candidate72ae413 then activated without +premature schema migration; its first saved-result read migrated safely to17. +An old16 Store opened before migration failed its actual Council handoff claim +at the database trigger; a fresh old16 Service failed `SchemaVersionError`. +Council version/claim remained unchanged. All owned groups were absent, all +runs terminal, and the temporary install was removed. Portable observed facts: +`evidence/r7-temporary-install-epoch-proof.json`. + +The strict installer regression passed in6.698s, then a separate full assertion +probe verified the strengthened explicit runtime scope and owned-group cleanup. +The controlled in-memory lower-version Store test remains labelled mechanics +only; the exact old payload gate above supplies actual installer provenance. +An independent readiness agent also froze72ae413 and demonstrated immutable +accepted old16 Service rejection at a real headless Council critic barrier, +identical full-table digest before/after the rejection, successful real fourth +fixture lead after restoring continuation, and all four PIDs absent. Root owns +recording that independent review alongside final integrated R6/native gates. +No production/root source/ledger, global AI settings, network or native generation +was changed. Earlier installer test failure was a test-only wrong status JSON +field (`claim` instead of `claimed_by`); corrected with exact SQL claim checks. +Final focused installed-epoch plus controlled-epoch rerun:2 tests passed +in7.976s with ResourceWarnings as errors. Bash all11 test files, generated +reference and diff checks passed before checkpointing this portable regression. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index 0448686..eb5cb6c 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -47,7 +47,7 @@ {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "complete", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-closure-2026-10-02.json", "checkpoint": "Accepted dfe9976/source equivalent f87060b and installed schema16. Initial native findings repaired; exact follow-up e046a174 clean/accepted. Full 504 tests/511.241s, two optional SDK skips/no errors/failures/unraisable; nine actual installed SDK tests pass/no skips. Public complete trial/evaluate/qualify/promote/new-run/held-out regression/rollback, shared call/active deadline and all terminal projection origins pass. Safe old-schema upgrade and idempotence/drift gates pass. Failed history retained; automatic experimentation and Jev stay off."}, - {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Baseline7130ee7/root32b94 passes554 full tests/645.034s, two optional SDK skips/no errors/failures/unraisable. Exact native audit44d0eecb explicitly rejected/failed22: unavailable probe ownership P1 and bytecode-invalidated check (66 tests ranOK, not accepted). Integrated repair093ed0f: strict36 tests on accepted3.12/11.250s and3.14/11.364s; root7/1.259s, matching source blobs, Bash/reference/diff pass. No captured identity means no protocol/output/group signals; kernel-confirmed direct-child fallback and retained exact zombie anchor preserve cleanup safely. Corrected no-bytecode invocation preserves integrity. Narrow native follow-up and final integrated/full/installed acceptance pending. Production remains accepted R5/schema16."}, + {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Ownership093 and EOF4f integrated with Council reconciliation0e/schema17. Rejected native44 and d235 findings and invalidated bytecode checks retained; strict40 tests each3.12/3.14, root34diagnostics/11.159s and8reconciliation/SDK21.651s pass. Plain-wheel all-schema regression red1/2.657s then green1/2.831s. Pristine exact trusted-check proof, narrow EOF native acceptance, final combined full and safe installed acceptance pending. User authorizes public GitHub core0.11 release in existing repository; Council disabled/native-unavailable."}, {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Isolated42979ba4/72ae4130/eee6db7: Council/epoch17/cancellation/comparison and actual immutable16-to17 temporary install proof pass. Active/recoverable16 defer unchanged, cancel/reconcile safely permits17; actual long-lived/fresh old16 clients reject Council mutation, ledger unchanged, new lead succeeds. Matched/heldout fixture comparison inconclusive; automatic use off, native quality/usage unknown. Native HTTPS attestation fails closed before generation; diagnostic paused for R6 ownership P1. Shared R6 authority/formatter reconciliation isolated on codex/council-integration; integrated full/native/installed acceptance remain open."}, {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} ], diff --git a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json index 2716976..9c3fdd3 100644 --- a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json @@ -2,7 +2,7 @@ "schema_version": 1, "work_package": "R6", "recorded_on": "2026-10-02", - "status": "ownership_repair_integrated_followup_final_full_install_pending", + "status": "ownership_and_eof_repair_integrated_final_acceptance_pending", "base_revision": "8f2c1c6a7d9edfe7cc3d78ecbc59ce105be636b2", "integrated_checkpoints": [ "dd709804e70e0791614319a9ae5c2597768d7318", @@ -14,8 +14,34 @@ "22a4ed8ac85b95861f4ba9725ca729388b7a9f49", "31a8a68b59db3d0af136c5c18cb67fe5088c02b0", "e7e9042cbf4d1416209cf39e0269f7fa85479f4c", - "093ed0f397289bffe9ea249c10796cb436027f00" + "093ed0f397289bffe9ea249c10796cb436027f00", + "4f01456189b38252a678938e7197ac2f12a6838e" ], + "eof_followup": { + "run_id": "d2355cd0-603d-4e6a-b1a2-b05b0e99d53a", + "target_revision": "8242eff68b24538aacc1b40b67b2e88e96b04d52", + "candidate_sha256": "349e55d96ae9ba541973b9f8eef7cfd2f1c7ccfb3f0f3e8ec2525a42edb203c9", + "state": "failed", "version": 22, "host_disposition": "reject", + "finding": "R6-EOF-exit-status", "severity": "medium", + "identity": "verified codex-cli0.159.2/gpt-6.1-sol/low/read_only", + "diff_and_bash": "passed_verified", + "affected_check": {"tests":56,"seconds":27.584,"runner_result":"OK","acceptance":"invalidated_by_undeclared_bytecode"}, + "bytecode_cause": "Controlled Python provider fixture children intentionally use a minimal environment and drop Python flags before importing core; parent _run_check already disabled bytecode. Repair uses -B only in controlled fixtures, with no native environment or integrity relaxation.", + "artifact_hashes_verified": [ + "23f319dbae10d44740863fed2ff6b806d22fedb5984edcc870c546ecf87d560e", + "53f2cb30cd9c51dfe737506dccde3f6cb77fabafd38ac95f237233a96a2148c8", + "ab3081b7057c2f398136ec0e4b489ccb6e0cca2caf1e5375b4459c17b4d48abd", + "5da3c924eac152a7a381ab259f7422688af34feb391834a4df4fb2be641e7b50" + ], + "repair_revision": "4f01456189b38252a678938e7197ac2f12a6838e", + "repair": "Nonreaping natural completion under original deadline before owned cleanup; preserve true version/auth exit status and retained ownership anchor; hung EOF stays bounded and unknown", + "red": {"tests":3,"failures":4,"seconds":0.745}, + "agent_gates": [{"python":"3.12","tests":40,"seconds":12.894},{"python":"3.14","tests":40,"seconds":13.398}], + "root_diagnostics": {"tests":34,"seconds":11.159,"failures":0,"errors":0,"resource_warnings":0}, + "root_council_sdk_reconciliation": {"tests":8,"seconds":21.651,"failures":0,"errors":0}, + "packaging_regression": {"red":{"tests":1,"seconds":2.657,"failures":1,"reason":"Council schema omitted from wheel"},"green":{"tests":1,"seconds":2.831,"failures":0,"errors":0},"scope":"actual plain installed wheel packages every source schema"}, + "final_native_full_install": "pending" + }, "contract": "SOL-REVIEW-FOLLOWUP R6/G3/G4; unchanged low-level MCP envelopes, fenced host disposition, Bash 3.2 and optional jq", "implemented": [ "Readable normal terminal commands; --json retains the versioned automation envelope", diff --git a/docs/plans/engineering-team/evidence/R7-NATIVE-BOUNDARY-BLOCKER.md b/docs/plans/engineering-team/evidence/R7-NATIVE-BOUNDARY-BLOCKER.md new file mode 100644 index 0000000..43d5eaf --- /dev/null +++ b/docs/plans/engineering-team/evidence/R7-NATIVE-BOUNDARY-BLOCKER.md @@ -0,0 +1,51 @@ +# Native Council boundary remains unavailable + +The October 2 isolated source gate did not make a generating native request. +MacOS 27.0 (26A5378n), `/usr/bin/sandbox-exec`, and the exact ChatGPT bundled +Codex CLI 0.159.2 were used. The last bounded metadata probe at 22:35 UTC sent +app-server `initialize`, `initialized`, and `model/list` over stdio, with copied +private subscription auth in a canonical private scratch HOME/CODEX_HOME. +All processes were owned-group cleaned and streams closed. The scratch/auth +copy was removed; no provider diagnostics or credentials are tracked. + +The exact default-deny profile's own role evidence and scratch were readable; +peer directories, ledger, private logs and artifacts were not. Binary/system +dependencies were limited to the exact executable, `/usr/lib` and +`/System/Library`, plus literal ancestors. Narrow managed-config metadata (not +config bytes), `/dev/urandom`, CFPreferences daemon/agent Mach services, exact +UID/daemon CFPreferences shared-memory names, and outbound TCP port 443 were +added. Initialization worked and model/list returned eight catalog entries. +Those entries do not prove a successful backend request. + +The last diagnostic variant also allowed these facilities, none committed as +native support: Mach services `com.apple.system.opendirectoryd.libinfo`, +`com.apple.mDNSResponder`, `com.apple.SystemConfiguration.configd`, +`com.apple.networkd`; outbound UDP port 53; `system-info net.link.addr`; exact +`/private/etc/hosts` read. It still reported +`failed to refresh available models: Connection failed: error sending request`. +There was no successful HTTPS response or HTTP status. Error classification: +network/backend attestation unavailable, underlying cause **unknown**. + +Earlier local sandbox logs demonstrated denied `system-info net.link.addr` +and `/private/etc/hosts` reads. Allowing them was not sufficient. DNS-related +Mach/UDP additions were diagnostic hypotheses, not a proven root cause. +Other denied notification/logging/user-preference/Info.plist/networkd-plist +operations were not shown necessary and were not broadly granted. No broad +filesystem, managed configuration, MCP or Mach-lookup exception was added. + +`verify_native_network` now fails capability-unavailable before any native role +launch. Doctor reports implemented partial mechanics but native-ready false. +Next: root-owned bounded non-generating backend attestation under the exact +frozen boundary (such as a genuinely successful native account/rate-limits +response), with the same peer/runtime/log/artifact/symlink/child denial probes. +Initialization, cached/fallback model lists, network error suppression, fixture +quality or a requested-only model ID must not lift this gate. + +The R6/Council reconciliation delegates version and bootstrap cleanup to the +shared bounded probe helper, refusing missing captured ownership before any +bootstrap RPC. Shared ownership repair +`093ed0f397289bffe9ea249c10796cb436027f00` is a separate explicit acceptance +dependency, not imported into this isolated Council checkpoint. Earlier local +cleanup observations above are not acceptance of the helper's ownership logic. +Root must accept and integrate that repair before new native probes; exact +backend attestation remains independently unavailable afterwards. diff --git a/docs/plans/engineering-team/evidence/r7-council-reconciliation-partial.json b/docs/plans/engineering-team/evidence/r7-council-reconciliation-partial.json new file mode 100644 index 0000000..21cda5c --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-council-reconciliation-partial.json @@ -0,0 +1,103 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "isolated_council_r6_source_reconciliation_offline_only", + "baseline_commit": "32b94d6", + "preserved_council_slices": ["42979ba", "72ae413", "eee6db7"], + "supported_schema_version": 17, + "automatic_enabled": false, + "native_ready": false, + "source_blob_sha256": { + "plugin/core/src/devsquad/diagnostics.py": "3ba04d754cc17ec6bf7cd32ded1b41d1972afba3d334952c5b87473c2b8bdfb4", + "plugin/core/src/devsquad/service.py": "b1d2ccb334f0c981e01e2a3fc5d8292ee7887f4c53d387ca6eb9c036707ba5c9", + "plugin/core/src/devsquad/store.py": "e94514dd50f398df26b4b4c9261eea9a4c4108d0da8ac40883ee55bb466fa88e", + "plugin/core/src/devsquad/cli.py": "e06d4a22a48a7c65bc53b16bf6e4b10b16d7318827788a8502608ea6a44875e5", + "plugin/core/src/devsquad/task_entry.py": "da29929711ea7eec268205855ead98cac9c368bac98e9296185f360a4574e72b", + "plugin/core/src/devsquad/council_isolation.py": "e5156c6d918b31fcdcc78b5b05dcdfa1a181392c54aba02ccbf92e53448a196e" + }, + "verification": { + "affected_gate": { + "command": "PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core python3 -m unittest test_council_contract test_council_entry test_council_runtime test_council_comparison test_council_epoch test_council_install_epoch test_council_isolation test_council_reconciliation test_terminal_ux test_diagnostics test_mcp test_handoff_store test_handoff_service test_task_entry test_router test_validation test_supervisor -q", + "python_version": "3.14.6", + "tests": 168, + "monotonic_seconds_reported_by_unittest": 148.693, + "errors": 0, + "failures": 0, + "skips": 2, + "skip_reason": "optional MCP SDK absent in this interpreter", + "status": "passed" + }, + "actual_official_sdk": { + "command": "PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core ACCEPTED_R5_PYTHON -m unittest test_mcp.OfficialSDKConformanceTest -q", + "python_version": "3.12.14", + "mcp_version": "2.2.0", + "tests": 2, + "monotonic_seconds_reported_by_unittest": 2.964, + "errors": 0, + "failures": 0, + "skips": 0, + "all_nine_tool_names_and_schema_envelopes_preserved": true, + "actual_stdio_disconnect_does_not_cancel_worker": true, + "status": "passed" + }, + "bash": {"command": "bash test/run.sh", "test_files": 11, "assertions": 227, "status": "passed"}, + "generated_reference": {"command": "python3 scripts/generate-core-reference.py --check", "status": "passed"}, + "diff": {"commands": ["git diff --check", "git diff --cached --check"], "status": "passed"} + }, + "retained_failed_gates": [ + { + "scope": "initial_expanded_headless_reconciliation_test", + "errors": 4, + "cause": "test wrongly expected inactive completed claims in the public active-claim view", + "repair": "assert exact authoritative claim rows; corrected live and expired crash tests pass" + }, + { + "scope": "first_final_affected_invocation", + "reported_tests": 169, + "monotonic_seconds_reported_by_unittest": 145.486, + "errors": 1, + "failures": 0, + "skips": 2, + "cause": "nonexistent test_reservation_crash module name produced unittest loader error", + "status": "failed_invocation_not_acceptance", + "repair": "corrected exact 168-test command above passed unchanged source" + } + ], + "shared_authority": { + "store_claim_parameters": ["initial_only", "terminal_decision"], + "marker_schema_version": 1, + "marker_reader": "Store.terminal_finish_decision", + "exact_live_intent_reuses_claim": true, + "exact_expired_intent_gets_fresh_fence": true, + "same_name_app_claim_not_adopted": true, + "guided_expiry_rejection_audited_and_recovered": true, + "submitted_crash_recovery_needs_no_new_claim": true, + "headless_expiry_retains_rejection_and_completes_saved_lead_without_new_worker": true, + "explicit_council_choice_inputs_round_trip_actual_cli": true, + "no_automatic_human_choice": true + }, + "preserved_evidence": [ + "r7-public-fixture-workflow-predeclaration.json", + "r7-public-fixture-workflow-comparison.json", + "r7-temporary-install-epoch-proof.json", + "R7-NATIVE-BOUNDARY-BLOCKER.md" + ], + "acceptance_dependencies": [ + { + "kind": "shared_owned_probe_cleanup", + "isolated_repair_commit": "093ed0f397289bffe9ea249c10796cb436027f00", + "root_integrated_commit": "8242eff68b24538aacc1b40b67b2e88e96b04d52", + "not_imported_into_this_council_checkpoint": true, + "independent_acceptance": "root_owned_pending" + }, + {"kind": "exact_default_deny_native_backend_attestation", "status": "unavailable_fail_closed_before_role_generation"}, + {"kind": "frozen_combined_full_native_and_installed_acceptance", "status": "root_owned_pending"} + ], + "limitations": [ + "No production/root source or ledger changes, network probes, native generations or full suite in this isolated reconciliation", + "Public fixtures establish mechanics only; native quality, escaped defects, rework and allowance remain unknown", + "Matched and held-out comparison remains inconclusive; no automatic use or R3 promotion authority", + "Historical temporary installer evidence names its original exact candidate; the reconciled candidate's installer regression also passed in the final affected command", + "Root resolves authoritative RESUME/backlog wording when importing this source-equivalent checkpoint" + ] +} diff --git a/docs/plans/engineering-team/evidence/r7-final-nongenerating-network-diagnostic.json b/docs/plans/engineering-team/evidence/r7-final-nongenerating-network-diagnostic.json new file mode 100644 index 0000000..be22501 --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-final-nongenerating-network-diagnostic.json @@ -0,0 +1,153 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "isolated_private_default_deny_nongenerating_diagnostic_only", + "status": "failed_backend_attestation_native_remains_unready", + "source_context_commit": "72ae4130190169096415436952f7c83498bd4d59", + "council_reconciliation_commit": "0e7dd319b56802329187007738970e26b5609b1d", + "production_profile_changed": false, + "production_native_ready": false, + "automatic_enabled": false, + "generating_requests": 0, + "paid_api_fallbacks": 0, + "native": { + "executable": "/Applications/ChatGPT.app/Contents/Resources/codex-cli/CodexCLI.app/Contents/MacOS/codex", + "executable_sha256": "50ac633af64851511f9bbc71032cdae7f1ba20b3234c189687d61ba846c354c5", + "verified_version": "codex-cli 0.159.2", + "sandbox_version_returncode": 0, + "sandbox_version_output_matches_expected": true, + "rpc_argv_suffix": ["--disable", "apps", "--disable", "plugins", "app-server", "--listen", "stdio://"], + "methods_sent": ["initialize", "initialized", "account/rateLimits/read"] + }, + "immutable_shared_helpers": { + "commit": "4f01456189b38252a678938e7197ac2f12a6838e", + "probe_process_git_blob": "f97ee3a1d19af929f096f0821df16b2cfed1182b", + "diagnostics_git_blob": "40c5849324b47b9014aee6f508116263375f155b", + "probe_process_sha256": "26668bfdd9a0e04cb98cf15859170ca0bcb9d8477377841b0e8888ff161d0e1f", + "diagnostics_sha256": "e8ef0b8315cf97c5586ee16ce1e3ec1be223991cccf54d9ee0aaea9aa6c18161", + "archive_mode": "0700", + "only_archived_source_files": ["diagnostics.py", "probe_process.py"], + "immediate_missing_identity_refusal": true, + "exclusive_popen_reaping_and_shared_cleanup": true, + "root_acceptance": "separate_root_owned_gate_not_claimed_by_this_diagnostic" + }, + "profile": { + "baseline_profile_sha256": "f363999d789c6bd75e0d579c5f9f25270a43c74e4e7c3a8a426db3b013048711", + "diagnostic_variant_profile_sha256": "70a1e58436302aec7d3f179fd8a468e0490bd175b95f1d3347bebe2745c393ed", + "actual_private_scratch_profile_sha256": "253c66503f0be900e9c35a089401c2addf317033d60464e88910d42f12305840", + "additions": [ + "(allow mach-lookup (global-name \"com.apple.system.opendirectoryd.libinfo\") (global-name \"com.apple.mDNSResponder\") (global-name \"com.apple.SystemConfiguration.configd\") (global-name \"com.apple.networkd\"))", + "(allow network-outbound (remote udp \"*:53\"))", + "(allow system-info (info-type \"net.link.addr\"))", + "(allow file-read* (literal \"/private/etc/hosts\"))", + "(allow network-outbound (literal \"/private/var/run/mDNSResponder\"))" + ], + "system_info_filter_syntax_authority": "/usr/share/sandbox/mDNSResponder.sb:151", + "cat_and_shell_execution_exceptions": "exact binaries added only to separate file-rule probes, never to native backend profile", + "broad_filesystem_or_mach_or_network_exceptions": false, + "managed_config_or_mcp_exceptions": false + }, + "actual_isolation_checks": { + "own_evidence_positive_control": {"returncode": 0, "output_bytes": 54}, + "peer_proposal": {"returncode": 1, "output_bytes": 0}, + "runtime_ledger": {"returncode": 1, "output_bytes": 0}, + "private_log": {"returncode": 1, "output_bytes": 0}, + "saved_artifact": {"returncode": 1, "output_bytes": 0}, + "own_directory_symlink_to_peer": {"returncode": 1, "output_bytes": 0}, + "inherited_child_runtime_ledger": {"returncode": 1, "output_bytes": 0}, + "unapproved_child_execution": {"returncode": 71, "output_bytes": 0}, + "all_eight_checks_preserved": true + }, + "sandbox_backend_request": { + "initialize_result": true, + "native_elapsed_seconds": 0.049, + "rpc_error_code": -32603, + "error_class": "network_request_failed", + "http_status": null, + "successful_native_response": false, + "owned_identity_captured": true, + "owned_group_cleanup_confirmed": true + }, + "single_unsandboxed_nongenerating_control": { + "reason": "separate_local_boundary_failure_from_backend_service_unavailable", + "initialize_result": true, + "native_elapsed_seconds": 0.49, + "non_error_rpc_result_received": true, + "strict_rate_limits_schema_valid": false, + "valid_window_count": 0, + "error_class": "invalid_or_empty_native_rate_limits", + "successful_native_response": false, + "owned_identity_captured": true, + "owned_group_cleanup_confirmed": true, + "schema_rejection_detail": "raw account/quota payload intentionally not retained; exact offending field is unknown, so no backend-success or service-outage conclusion" + }, + "backend_validator": { + "plain_dictionary_or_cached_catalog_is_insufficient": true, + "required_top_level_keys": ["rateLimits", "rateLimitsByLimitId"], + "allowed_snapshot_keys": ["limitId", "limitName", "primary", "secondary", "credits", "planType"], + "required_snapshot_keys": ["primary", "secondary"], + "window_fields": ["usedPercent", "windowDurationMins", "resetsAt"], + "window_requirements": "finite nonboolean usage 0..100; positive integer duration; integer future reset; at least one valid window", + "credits_requirements": "exact hasCredits/unlimited boolean and bounded nullable balance shape", + "strings_and_map_keys": "bounded nonempty strings; optional nullable fields remain nullable", + "successful_schema_validated_receipts": 0 + }, + "bounded_execution": { + "paired_backend_deadline_seconds": 12, + "sandbox_backend_attempts": 1, + "unsandboxed_backend_controls": 1, + "further_network_attempts": 0, + "python_version": "3.12.14", + "resource_warnings": "error", + "bytecode_writes": "disabled by -B and sys.dont_write_bytecode", + "native_pid_absence_independently_checked_after_cleanup": true + }, + "privacy_and_preservation": { + "only_auth_json_copied": true, + "private_scratch_mode": "0700", + "temporary_auth_copy_mode": "0600", + "temporary_auth_copy_removed": true, + "original_auth_unchanged": true, + "original_diagnostic_script_sha256_unchanged": "e6f2cb9cdc371c595cc0826a99f7614e65a1aa797ff4c8c883d9b1b183a7dc23", + "original_diagnostic_result_sha256_unchanged": "6f334467e4dc0beb3a57bfc395427db24e42fa73587dbddbf642ef5480ab8ae3", + "raw_provider_messages_auth_account_and_quota_values_tracked": false, + "final_private_script_sha256": "1480f38dd83fe45a30bf79ab92da387fa01a34f5bacb4efae9c7d811a4b08e90", + "final_private_redacted_receipt_sha256": "c125c3283e994f7c7dad8322e9e0889008f46a7c7935d3c0164e7891ce40d8c1" + }, + "retained_early_aborts": [ + { + "helper_commit": "093ed0f397289bffe9ea249c10796cb436027f00", + "stage": "sandbox_version_gate_before_file_checks_or_backend_rpc", + "returncode": null, + "cause": "exact return code not recorded; do not attribute it to the known EOF helper defect without evidence", + "private_redacted_receipt_sha256": "c8ef5f89b0fcfdd6a7831b01044c84a339bf030bf8f83cde1365a3130e83d2a9" + }, + { + "helper_commit": "4f01456189b38252a678938e7197ac2f12a6838e", + "stage": "sandbox_version_gate_before_file_checks_or_backend_rpc", + "returncode": 65, + "cause": "bare net.link.addr token in system-info rule was an unbound variable; exact no-auth/no-native compiler check proved syntax defect, corrected with OS-defined info-type filter", + "private_redacted_receipt_sha256": "fa39261aba4fbe4f56112cd39f991f7cceb1801558405190f5bffc1e8d0f27c9" + }, + { + "stage": "first_no_auth_no_native_compiler_check_invocation", + "cause": "command helper correctly refused replacing the frozen executable before profile construction; corrected invocation retained frozen profile then used a denied true executable solely for parser diagnostics", + "backend_requests": 0 + } + ], + "scoped_previous_pid_log_query": { + "pid": 63678, + "info": true, + "debug": true, + "predicate_scope": "only exact codex PID sandbox deny messages, excluding the log query process", + "matching_denials": 0, + "missing_facility_inference": "none; no specific additional OS facility has been demonstrated necessary or sufficient" + }, + "next_action": "Stop network attempts. Root may investigate the exact local boundary and validator contract separately; Council remains fail-closed before native role generation. This diagnostic is not production readiness, a minimum-boundary proof, native quality evaluation, or release acceptance.", + "verification": { + "bash": {"command": "bash test/run.sh", "status": "passed", "test_files": 11, "assertions": 227}, + "generated_reference": {"command": "python3 -B scripts/generate-core-reference.py --check", "status": "passed"}, + "diff": {"command": "git diff --check", "status": "passed"}, + "json_parse": {"command": "python3 -B -m json.tool docs/plans/engineering-team/evidence/r7-final-nongenerating-network-diagnostic.json", "status": "passed"} + } +} diff --git a/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-comparison.json b/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-comparison.json new file mode 100644 index 0000000..269f4de --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-comparison.json @@ -0,0 +1,204 @@ +{ + "automatic_enabled": false, + "cases": [ + { + "arms": { + "control": { + "accepted_quality": null, + "actual_prompt_sha256": [ + "6841c7ab565bb86c35c4c40c47ca7831c2119b8122e788a78826c41eeed12982" + ], + "brief_sha256": null, + "candidate": { + "base_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc", + "sha256": "ef2f034431385621374f5e97428107e742c8f583b6bbcb0ba4174452e9ea0aaa", + "target_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc" + }, + "escaped_defects": null, + "execution_elapsed_ms": 3640, + "host_usage": null, + "identity_scope": "all_fixture", + "input_contract_sha256": "3e4ee466f5141b2cd2bcea35e14adf1354cf91448cd3743855b1c2349d756d72", + "lead_disposition": "accept", + "native_model_requests": null, + "native_quota": null, + "native_usage": [ + { + "input_tokens": null, + "output_tokens": null, + "source": "unavailable", + "total_tokens": null + } + ], + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "receipt": { + "artifact_id": "a6a606ef-f798-4684-936b-a6dded2cbadf", + "sha256": "09275379766dc1ab52e26c69a33028d9d4f4a3e0311154e82fd85a2c56469595" + }, + "rework": null, + "run_id": "76b924c1-583a-4fe0-bb3f-f27088e1ea3e", + "runtime_package_sha256": "b456841b6c89fbbfbc0ea488c1a736d3df9056c3551857dca08432b97710cbd6", + "state": "succeeded", + "versions": { + "policy_sha256": "076ee3ae42e47cdfa86eee7a27a4ade567215ba9f6d1a978a2f3585332f5a337", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "workflows.py", + "prompt_module_sha256": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "task_sha256": "c84e2441b70f0fed5a3c86ade31a2a5f811fb2a0ec2f5ec79c6fbad48220f6dd" + }, + "worker_invocations": 1 + }, + "council": { + "accepted_quality": null, + "actual_prompt_sha256": [ + "8b2359053fd3dba49b621c09810cb7dc56889dbcfca0faf0a7f03529e70eb6fc", + "5afd1556f9e7e539759c8cf080042e4221d5556ea0930e3d0b36c9d92bc297df", + "aae30337bfb1e911e000cb354a9968a68483f439181e16a1a264cdbcfae7a9c3" + ], + "brief_sha256": "01a67fd5ba56856a1e58ac28494af09d74c41251c3872f6065d51922ffe7dc1d", + "candidate": { + "base_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc", + "sha256": "ef2f034431385621374f5e97428107e742c8f583b6bbcb0ba4174452e9ea0aaa", + "target_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc" + }, + "escaped_defects": null, + "execution_elapsed_ms": 6087, + "host_usage": null, + "identity_scope": "all_fixture", + "input_contract_sha256": "3e4ee466f5141b2cd2bcea35e14adf1354cf91448cd3743855b1c2349d756d72", + "lead_disposition": "accept", + "native_model_requests": null, + "native_quota": null, + "native_usage": [ + null, + null, + null + ], + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "receipt": { + "artifact_id": "9192542d-a6bf-4886-9360-bf05ad76d2f4", + "sha256": "ba7fbd67466fce66795a148cf9724fa40db06702ad858bdb84c44162390be47f" + }, + "rework": null, + "run_id": "7f01a92a-5458-44f6-875d-124aab6bda54", + "runtime_package_sha256": "b456841b6c89fbbfbc0ea488c1a736d3df9056c3551857dca08432b97710cbd6", + "state": "succeeded", + "versions": { + "policy_sha256": "4fac321b5ec269b8c59f0113771c510fd4080dcc8a102e693ba128351388a46a", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "council.py", + "prompt_module_sha256": "9646b5a6455762c766a2d54b72618ad15ce0cd662309ce61c1ae045e9be8c81f", + "task_sha256": "2d82a04e0904a7c5601f2171a90e2e9976ead07c0810a0d8880e43e9cfc0d1ce" + }, + "worker_invocations": 3 + } + }, + "id": "retry-key", + "split": "matched" + }, + { + "arms": { + "control": { + "accepted_quality": null, + "actual_prompt_sha256": [ + "b5d07a1bff9c3b3c0e28c364b959d454b1fac07463754a1d12439e593807f61f" + ], + "brief_sha256": null, + "candidate": { + "base_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc", + "sha256": "ef2f034431385621374f5e97428107e742c8f583b6bbcb0ba4174452e9ea0aaa", + "target_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc" + }, + "escaped_defects": null, + "execution_elapsed_ms": 3531, + "host_usage": null, + "identity_scope": "all_fixture", + "input_contract_sha256": "c6bc137c2c1288edda333f7a8ba262e7774723efc60c4b0989ec3ef33a10e3ae", + "lead_disposition": "accept", + "native_model_requests": null, + "native_quota": null, + "native_usage": [ + { + "input_tokens": null, + "output_tokens": null, + "source": "unavailable", + "total_tokens": null + } + ], + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "receipt": { + "artifact_id": "d3cb199a-f927-4e11-a9a2-915ba5c4eda4", + "sha256": "57901f4791141741a75d9bf3b6079f49f44ed782cd828eb8f15466767d08038a" + }, + "rework": null, + "run_id": "e5342195-90e6-47d7-b055-529e4013ef41", + "runtime_package_sha256": "b456841b6c89fbbfbc0ea488c1a736d3df9056c3551857dca08432b97710cbd6", + "state": "succeeded", + "versions": { + "policy_sha256": "076ee3ae42e47cdfa86eee7a27a4ade567215ba9f6d1a978a2f3585332f5a337", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "workflows.py", + "prompt_module_sha256": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "task_sha256": "fd5adc9eb398a381be9cb14900b3606a95c5f2962176aa541d3b873db38175a2" + }, + "worker_invocations": 1 + }, + "council": { + "accepted_quality": null, + "actual_prompt_sha256": [ + "1b5b9192d21b912bd78ad05e0ee0c368df2edb2c4926f561c485c2a8cd9e5259", + "add9d2c081540e13c95a0db62c41e4f016060c509ea6b06607cada2b1aea8e01", + "ebb71d34afc939968cdfd51948e4bfb0b3bbe482834e787fd6066ce0a895b354" + ], + "brief_sha256": "632758d5386d411ebe7021ce1d4cdf78efa1fde57a69f12737b63b781630aacd", + "candidate": { + "base_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc", + "sha256": "ef2f034431385621374f5e97428107e742c8f583b6bbcb0ba4174452e9ea0aaa", + "target_oid": "e9e75b6801a87b6d844595e1a3356847a15346dc" + }, + "escaped_defects": null, + "execution_elapsed_ms": 4528, + "host_usage": null, + "identity_scope": "all_fixture", + "input_contract_sha256": "c6bc137c2c1288edda333f7a8ba262e7774723efc60c4b0989ec3ef33a10e3ae", + "lead_disposition": "accept", + "native_model_requests": null, + "native_quota": null, + "native_usage": [ + null, + null, + null + ], + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "receipt": { + "artifact_id": "b0978f78-a739-4605-aa52-d24e5ca52eee", + "sha256": "27b43840c0fcdb10777f0bd970b7d9015cecb07fa8fd07331e36178816a6d463" + }, + "rework": null, + "run_id": "fd52dd7d-5e4d-45c5-9b0c-e0977c9113de", + "runtime_package_sha256": "b456841b6c89fbbfbc0ea488c1a736d3df9056c3551857dca08432b97710cbd6", + "state": "succeeded", + "versions": { + "policy_sha256": "4fac321b5ec269b8c59f0113771c510fd4080dcc8a102e693ba128351388a46a", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "council.py", + "prompt_module_sha256": "9646b5a6455762c766a2d54b72618ad15ce0cd662309ce61c1ae045e9be8c81f", + "task_sha256": "6a1527d0de63509dced266a249dfb3472fccf851de82bd51e71cc44fd89bbefd" + }, + "worker_invocations": 3 + } + }, + "id": "retry-expiry", + "split": "heldout" + } + ], + "conclusion": "inconclusive", + "limitations": [ + "Controlled process mechanics only", + "No independent native accepted-quality/escaped-defect/rework observations", + "Workflow prompts and invocation count differ; this is not R3 single-binding eligibility" + ], + "predeclaration_sha256": "5f96d3946d2bd32a0aed16298533998d308da19e68e600daf300806ee2b7a43f", + "quality_benefit_supported": false, + "schema_version": 1 +} diff --git a/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-predeclaration.json b/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-predeclaration.json new file mode 100644 index 0000000..2b94f96 --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-public-fixture-workflow-predeclaration.json @@ -0,0 +1,52 @@ +{ + "automatic_enabled": false, + "cases": [ + { + "control": { + "policy_sha256": "076ee3ae42e47cdfa86eee7a27a4ade567215ba9f6d1a978a2f3585332f5a337", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "workflows.py", + "prompt_module_sha256": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "task_sha256": "c84e2441b70f0fed5a3c86ade31a2a5f811fb2a0ec2f5ec79c6fbad48220f6dd" + }, + "council": { + "policy_sha256": "4fac321b5ec269b8c59f0113771c510fd4080dcc8a102e693ba128351388a46a", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "council.py", + "prompt_module_sha256": "9646b5a6455762c766a2d54b72618ad15ce0cd662309ce61c1ae045e9be8c81f", + "task_sha256": "2d82a04e0904a7c5601f2171a90e2e9976ead07c0810a0d8880e43e9cfc0d1ce" + }, + "id": "retry-key", + "input_contract_sha256": "3e4ee466f5141b2cd2bcea35e14adf1354cf91448cd3743855b1c2349d756d72", + "split": "matched" + }, + { + "control": { + "policy_sha256": "076ee3ae42e47cdfa86eee7a27a4ade567215ba9f6d1a978a2f3585332f5a337", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "workflows.py", + "prompt_module_sha256": "6e0be346a78a50927913762fe78e989a3b73f56f84808fdbda3886de7edf00ef", + "task_sha256": "fd5adc9eb398a381be9cb14900b3606a95c5f2962176aa541d3b873db38175a2" + }, + "council": { + "policy_sha256": "4fac321b5ec269b8c59f0113771c510fd4080dcc8a102e693ba128351388a46a", + "profiles_sha256": "24e7c79cacb1b896f84fe838a5f22f52ce2bdbbfd33cde4dd2760c6a8c4e333b", + "prompt_module": "council.py", + "prompt_module_sha256": "9646b5a6455762c766a2d54b72618ad15ce0cd662309ce61c1ae045e9be8c81f", + "task_sha256": "6a1527d0de63509dced266a249dfb3472fccf851de82bd51e71cc44fd89bbefd" + }, + "id": "retry-expiry", + "input_contract_sha256": "c6bc137c2c1288edda333f7a8ba262e7774723efc60c4b0989ec3ef33a10e3ae", + "split": "heldout" + } + ], + "created_at": "2026-10-02T23:01:18.274582+00:00", + "decision_rule": "inconclusive_without_independent_native_quality_evidence", + "quality_gate": { + "accepted_quality": "unmeasured", + "escaped_defects": "unmeasured", + "rework": "unmeasured" + }, + "schema_version": 1, + "scope": "public_fixture_mechanics" +} diff --git a/docs/plans/engineering-team/evidence/r7-temporary-install-epoch-proof.json b/docs/plans/engineering-team/evidence/r7-temporary-install-epoch-proof.json new file mode 100644 index 0000000..49b8acd --- /dev/null +++ b/docs/plans/engineering-team/evidence/r7-temporary-install-epoch-proof.json @@ -0,0 +1,30 @@ +{ + "schema_version": 1, + "scope": "actual_temporary_core_only_installer_and_public_fixture_processes", + "candidate_commit": "72ae4130190169096415436952f7c83498bd4d59", + "candidate_source_sha256": "4c6f0d92e6576abcf2f64044a0321cf9cd8c8e0c808d7923e0f9b7e97b8b4cab", + "old_commit": "bf3de0867484552d354e6e8b6ba835f31a93aeb3", + "old_source_sha256": "90f1e87fb9acf7756dad46780672de9b84fe75e3631322857a310f089280345e", + "old_package_sha256": "e03bf3a2e362b3aff6d0a5dd00bd837f15afb216d4c58ac1d9c776c4d8f09c67", + "old_schema": 16, + "new_schema": 17, + "active_deferral": true, + "recoverable_deferral": true, + "selector_and_launcher_preserved": true, + "old_active_cancelled": true, + "old_recoverable_reconciled": true, + "long_lived_old_claim_rejected": true, + "fresh_old_service_claim_rejected": true, + "council_claim_unchanged": true, + "owned_process_groups_gone": true, + "temporary_install_removed": true, + "mcp_installed_in_temporary_release": false, + "native_generation": false, + "regression_test": "test/core/test_council_install_epoch.py", + "limitations": [ + "This is a temporary core-only installation, not a production upgrade or MCP SDK lifecycle gate", + "The exact archived core matches accepted immutable R5 source and code-package digests; no version-rewritten Store was used", + "Public fixture mechanics do not prove native quality or network readiness", + "Root must reconcile R6 shared code and repeat integrated acceptance before production activation" + ] +} diff --git a/plugin/.claude-plugin/plugin.json b/plugin/.claude-plugin/plugin.json index 97eb5db..bd2b19d 100644 --- a/plugin/.claude-plugin/plugin.json +++ b/plugin/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "devsquad", - "version": "0.10.0", + "version": "0.11.0", "description": "Engineering Manager that coordinates AI coding agents through enforced delegation", "author": { "name": "Dikshant Joshi", @@ -17,4 +17,4 @@ "multi-agent", "token-management" ] -} \ No newline at end of file +} diff --git a/plugin/core/pyproject.toml b/plugin/core/pyproject.toml index 9d4222b..a498814 100644 --- a/plugin/core/pyproject.toml +++ b/plugin/core/pyproject.toml @@ -29,7 +29,7 @@ where = ["src"] "share/devsquad/adapters/grok" = ["adapters/grok/adapter.json"] "share/devsquad/adapters/claude" = ["adapters/claude/adapter.json"] "share/devsquad/adapters" = ["adapters/classification-policy.conf"] -"share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/check-result.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/profiles.schema.json", "schemas/review-result.schema.json", "schemas/task.schema.json"] +"share/devsquad/schemas" = ["schemas/adapter.schema.json", "schemas/check-result.schema.json", "schemas/council.schema.json", "schemas/execution-identity.schema.json", "schemas/launch-spec.schema.json", "schemas/normalized-result.schema.json", "schemas/policy.schema.json", "schemas/profile.schema.json", "schemas/profiles.schema.json", "schemas/review-result.schema.json", "schemas/task.schema.json"] "share/devsquad/profiles" = ["profiles/templates.json"] "share/devsquad/integrations/codex" = ["integrations/codex/registration.json"] "share/devsquad/integrations/claude-code" = ["integrations/claude-code/registration.json"] diff --git a/plugin/core/schemas/council.schema.json b/plugin/core/schemas/council.schema.json new file mode 100644 index 0000000..e2717f6 --- /dev/null +++ b/plugin/core/schemas/council.schema.json @@ -0,0 +1,17 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://devsquad.local/schemas/council-v1.json", + "type": "object", "additionalProperties": false, + "required": ["schema_version", "enabled", "automatic", "reason", "min_valid_proposals", "required_critics", "max_invocations", "seed", "evidence", "rubric"], + "properties": { + "schema_version": {"const": 1}, "enabled": {"const": true}, "automatic": {"const": false}, + "reason": {"type": "string", "minLength": 1, "maxLength": 16000}, + "min_valid_proposals": {"const": 2}, "required_critics": {"const": 1}, + "max_invocations": {"type": "integer", "minimum": 3, "maximum": 16}, + "seed": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "evidence": {"type": "array", "maxItems": 32, "items": {"type": "object", "additionalProperties": false, + "required": ["artifact_id", "sha256"], "properties": {"artifact_id": {"type": "string", "minLength": 1, "maxLength": 16000}, "sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}}}}, + "rubric": {"type": "array", "minItems": 1, "maxItems": 32, "items": {"type": "object", "additionalProperties": false, + "required": ["id", "description"], "properties": {"id": {"type": "string", "minLength": 1, "maxLength": 16000}, "description": {"type": "string", "minLength": 1, "maxLength": 16000}}}} + } +} diff --git a/plugin/core/schemas/policy.schema.json b/plugin/core/schemas/policy.schema.json index 76884f2..fd91534 100644 --- a/plugin/core/schemas/policy.schema.json +++ b/plugin/core/schemas/policy.schema.json @@ -4,7 +4,7 @@ "required":["schema_version","id","version","roles","task_classes","require_different_model_for_review","account_pools","experiment_budget"], "properties":{ "schema_version":{"const":1},"id":{"type":"string","minLength":1},"version":{"type":"integer","minimum":1}, - "roles":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"type":"array","minItems":1,"items":{"$ref":"#/$defs/candidate"}}}, + "roles":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher","proposer_a","proposer_b","critic"]},"additionalProperties":{"type":"array","minItems":1,"items":{"$ref":"#/$defs/candidate"}}}, "task_classes":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"enum":["unvalidated","trial","proven","suspended"]}},"require_different_model_for_review":{"type":"boolean"},"prefer_different_harness_for_review":{"type":"boolean"}, "account_pools":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"$ref":"#/$defs/account_pool"}},"experiment_budget":{"type":"object","propertyNames":{"minLength":1},"additionalProperties":{"type":"integer","minimum":0}},"decision_helper":{"oneOf":[{"type":"object","additionalProperties":false,"required":["schema_version","mode"],"properties":{"schema_version":{"const":1},"mode":{"const":"off"}}},{"$ref":"#/$defs/decision_helper_enabled"}]} }, diff --git a/plugin/core/schemas/task.schema.json b/plugin/core/schemas/task.schema.json index b779769..ef50f63 100644 --- a/plugin/core/schemas/task.schema.json +++ b/plugin/core/schemas/task.schema.json @@ -3,13 +3,14 @@ "type": "object", "additionalProperties": false, "required": ["schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin"], "properties": { - "schema_version": {"const": 1}, "workflow": {"enum": ["branch-review", "issue-delivery"]}, + "schema_version": {"const": 1}, "workflow": {"enum": ["branch-review", "issue-delivery", "council-decision"]}, "goal": {"type": "string", "minLength": 1}, "task_class": {"type": "string", "minLength": 1}, "project": {"$ref": "#/$defs/project"}, "acceptance": {"type": "array", "minItems": 1, "maxItems": 100, "items": {"$ref": "#/$defs/acceptance"}}, "checks": {"type": "array", "maxItems": 16, "items": {"$ref": "#/$defs/check"}}, "scope": {"$ref": "#/$defs/scope"}, "lead": {"$ref": "#/$defs/lead"}, "routing": {"$ref": "#/$defs/routing"}, "budget": {"$ref": "#/$defs/budget"}, - "origin": {"$ref": "#/$defs/origin"}, "review": {"$ref": "#/$defs/review"} + "origin": {"$ref": "#/$defs/origin"}, "review": {"$ref": "#/$defs/review"}, "council": {"$ref": "council.schema.json"} }, + "allOf": [{"if": {"properties": {"workflow": {"const": "council-decision"}}}, "then": {"required": ["council"], "properties": {"budget": {"properties": {"max_revisions": {"const": 0}}}, "scope": {"properties": {"write_paths": {"maxItems": 0}}}}, "not": {"required": ["review"]}}, "else": {"not": {"required": ["council"]}}}], "$defs": { "project": {"type":"object","additionalProperties":false,"required":["repo_path","base_ref","target_ref"],"properties":{"repo_path":{"type":"string","pattern":"^/"},"base_ref":{"type":"string","minLength":1},"target_ref":{"type":"string","minLength":1}}}, "acceptance": {"type":"object","additionalProperties":false,"required":["id","description","evidence_kind"],"properties":{"id":{"type":"string","minLength":1},"description":{"type":"string","minLength":1},"evidence_kind":{"enum":["review","check","artifact","host"]}}}, @@ -17,7 +18,7 @@ "scope": {"type":"object","additionalProperties":false,"required":["read_paths","write_paths"],"properties":{"read_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}},"write_paths":{"type":"array","maxItems":256,"uniqueItems":true,"items":{"type":"string","minLength":1}}}}, "lead": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["host","headless"]}}}, "override": {"type":"object","additionalProperties":false,"required":["profile_id"],"properties":{"profile_id":{"type":"string","minLength":1},"fallback":{"enum":["none","policy"]}}}, - "routing": {"type":"object","additionalProperties":false,"properties":{"profiles_file":{"type":"string","minLength":1},"policy_file":{"type":"string","minLength":1},"profiles":{"type":"object"},"policy":{"type":"object"},"overrides":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher"]},"additionalProperties":{"$ref":"#/$defs/override"}}},"oneOf":[{"required":["profiles_file","policy_file"],"not":{"anyOf":[{"required":["profiles"]},{"required":["policy"]}]}},{"required":["profiles","policy"],"not":{"anyOf":[{"required":["profiles_file"]},{"required":["policy_file"]}]}}]}, + "routing": {"type":"object","additionalProperties":false,"properties":{"profiles_file":{"type":"string","minLength":1},"policy_file":{"type":"string","minLength":1},"profiles":{"type":"object"},"policy":{"type":"object"},"overrides":{"type":"object","propertyNames":{"enum":["implementer","reviewer","lead","researcher","proposer_a","proposer_b","critic"]},"additionalProperties":{"$ref":"#/$defs/override"}}},"oneOf":[{"required":["profiles_file","policy_file"],"not":{"anyOf":[{"required":["profiles"]},{"required":["policy"]}]}},{"required":["profiles","policy"],"not":{"anyOf":[{"required":["profiles_file"]},{"required":["policy_file"]}]}}]}, "budget": {"type":"object","additionalProperties":false,"required":["wall_seconds","max_worker_invocations","max_revisions","max_fallbacks_per_step"],"properties":{"wall_seconds":{"type":"integer","minimum":1},"max_worker_invocations":{"type":"integer","minimum":1},"max_revisions":{"type":"integer","minimum":0},"max_fallbacks_per_step":{"type":"integer","minimum":0}}}, "origin": {"type":"object","additionalProperties":false,"required":["surface"],"properties":{"surface":{"type":"string","minLength":1},"session_ref":{"type":"string","minLength":1}}}, "review": {"type":"object","additionalProperties":false,"required":["mode"],"properties":{"mode":{"enum":["standard","adversarial"]},"focus":{"type":"string","minLength":1}}} diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index bd192bd..b2e2a13 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -36,7 +36,7 @@ "cancelled": 4, } WAIT_ACTIVE_STATES = {"queued", "running", "cancelling"} -HUMAN_COMMANDS = {"review", "fix", "status", "result", "doctor", "setup", "finish", "resume", "cancel"} +HUMAN_COMMANDS = {"review", "fix", "council", "council-finish", "status", "result", "doctor", "setup", "finish", "resume", "cancel"} def manifests() -> list[tuple[Path, AdapterManifest]]: @@ -327,6 +327,61 @@ def command_fix(args: argparse.Namespace) -> tuple[dict, int]: return _command_normal_entry(args, workflow="issue-delivery") +def command_council(args: argparse.Namespace) -> tuple[dict, int]: + from .council_task_entry import build_council_task + if args.dry_run and args.wait: + raise ContractError("--wait cannot be combined with --dry-run") + repo = resolve_repository(args.project_dir) + identities = discover_codex_identity(repo, requested_effort=args.effort, runtime=Path(args.runtime_dir), all_models=True) + if args.model: + if len(args.model) != 3 or len(set(args.model)) != 3: + raise ContractError("Council --model must name exactly three distinct model IDs in proposer_a/proposer_b/critic order") + catalog = {value["model_id"]: value for value in identities} + if any(model not in catalog for model in args.model): + raise ContractError("Council pinned model is not available with verified effort metadata") + identities = [catalog[model] for model in args.model] + else: + # Prefer available catalog cross-family IDs; catalog entitlement is not + # tested quality qualification. The same Codex subscription harness + # with three entitled distinct IDs remains a valid bounded baseline. + chosen = [] + remaining = list(identities) + while remaining and len(chosen) < 3: + families = {identity["model_family"] for identity in chosen} + remaining.sort(key=lambda identity: identity["model_family"] in families) + chosen.append(remaining.pop(0)) + identities = chosen + evidence = [] + for value in args.evidence or []: + artifact_id, separator, sha256 = value.rpartition(":") + if not separator: + raise ContractError("Council --evidence must be ARTIFACT_ID:SHA256") + evidence.append({"artifact_id": artifact_id, "sha256": sha256}) + rubric = None + if args.criterion: + rubric = [] + for value in args.criterion: + criterion_id, separator, description = value.partition("=") + if not separator: + raise ContractError("Council --criterion must be ID=DESCRIPTION") + rubric.append({"id": criterion_id, "description": description}) + task, summary = build_council_task(project_dir=repo, goal=args.question, identities=identities, + base_ref=args.base, target_ref=args.target, read_paths=args.read_path or (), checks=parse_checks(args.check), + lead_mode=args.lead, max_invocations=args.max_invocations, evidence=evidence, rubric=rubric) + key = args.idempotency_key or f"normal-council-{summary['task_sha256']}" + if args.dry_run: + return envelope(data=_normal_entry_result(summary, key, None)), 0 + service = _service(args) + started = service.start(task, key) + run, code = _wait_for_run(service, started) if args.wait else (envelope(data=started), 0) + service_data = run["data"] if args.json else _display_status(service, run["data"]) + return envelope(data=_normal_entry_result(summary, key, service_data)), code + + +def command_council_finish(args: argparse.Namespace) -> tuple[dict, int]: + return command_finish(args) + + def command_capacity_observe(args: argparse.Namespace) -> tuple[dict, int]: observation = _read_json(args.file, "capacity observation file") return envelope(data=_service(args).capacity_observe(observation)), 0 @@ -423,6 +478,8 @@ def command_finish(args: argparse.Namespace) -> tuple[dict, int]: service = _service(args) return envelope(data=service.finish( _selected_run(args, service), args.disposition, args.reason, + chosen=args.choose, supported_claims=args.supported_claim, + discarded_alternatives=args.discarded_alternative, validation=args.validation, )), 0 @@ -500,6 +557,39 @@ def parser() -> argparse.ArgumentParser: cmd.add_argument("--stderr-file", required=True) cmd.set_defaults(func=fn) runtime_default = os.environ.get("DEVSQUAD_RUNTIME_DIR", str(Path.home() / ".devsquad" / "runtime")) + council = sub.add_parser("council", help="manual read-only Council decision, automatic triggering off") + council.add_argument("question") + council.add_argument("--project-dir", default=str(Path.cwd())) + council.add_argument("--base", default="HEAD") + council.add_argument("--target", default="HEAD") + council.add_argument("--read-path", action="append") + council.add_argument("--evidence", action="append", help="saved ARTIFACT_ID:SHA256") + council.add_argument("--criterion", action="append", help="ID=DESCRIPTION; freezes the complete rubric") + council.add_argument("--check", action="append") + council.add_argument("--model", action="append", help="exact model IDs, repeated three times") + council.add_argument("--effort") + council.add_argument("--lead", choices=("headless", "host"), default="headless") + council.add_argument("--max-invocations", type=int, default=4) + council.add_argument("--idempotency-key") + council.add_argument("--dry-run", action="store_true") + council.add_argument("--wait", action="store_true") + council.add_argument("--json", action="store_true") + council.add_argument("--runtime-dir", default=runtime_default) + council.set_defaults(func=command_council) + council_finish = sub.add_parser("council-finish", help="explicit host Council choice without decision JSON") + council_finish.add_argument("run", nargs="?") + council_disposition = council_finish.add_mutually_exclusive_group(required=True) + council_disposition.add_argument("--accept", dest="disposition", action="store_const", const="accept") + council_disposition.add_argument("--reject", dest="disposition", action="store_const", const="reject") + council_finish.add_argument("--reason", required=True) + council_finish.add_argument("--choose", choices=("A", "B", "synthesis"), required=True) + council_finish.add_argument("--supported-claim", action="append") + council_finish.add_argument("--discarded-alternative", action="append") + council_finish.add_argument("--validation", required=True) + council_finish.add_argument("--json", action="store_true") + council_finish.add_argument("--project-dir", default=str(Path.cwd())) + council_finish.add_argument("--runtime-dir", default=runtime_default) + council_finish.set_defaults(func=command_council_finish) review = sub.add_parser( "review", help="start an exact-commit Codex branch review without task JSON", @@ -567,6 +657,10 @@ def parser() -> argparse.ArgumentParser: for value in ("accept", "reject", "revise"): disposition.add_argument(f"--{value}", dest="disposition", action="store_const", const=value) finish.add_argument("--reason", required=True) + finish.add_argument("--choose", choices=("A", "B", "synthesis"), help="required for Council, never inferred") + finish.add_argument("--supported-claim", action="append", help="Council supported claim, repeat as needed") + finish.add_argument("--discarded-alternative", action="append", help="Council discarded alternative, repeat as needed") + finish.add_argument("--validation", help="required Council objective validation and remaining uncertainty") finish.add_argument("--project-dir", default=str(Path.cwd())) finish.add_argument("--json", action="store_true") finish.add_argument("--runtime-dir", default=runtime_default) @@ -660,12 +754,22 @@ def _next_command(data: dict[str, Any]) -> str: if action == "claim_handoff": pending = (data.get("handoff_view") or {}).get("pending_finish") if pending: - return f"squad finish {run_id} --{pending['disposition']} --reason={shlex.quote(pending['reason'])}" + command = f"squad finish {run_id} --{pending['disposition']} --reason={shlex.quote(pending['reason'])}" + choice = pending.get("council_choice") + if choice is not None: + command += f" --choose={shlex.quote(choice['chosen'])} --validation={shlex.quote(choice['validation'])}" + for value in choice["supported_claims"]: + command += f" --supported-claim={shlex.quote(value)}" + for value in choice["discarded_alternatives"]: + command += f" --discarded-alternative={shlex.quote(value)}" + return command owner = (data.get("handoff") or {}).get("claimed_by") if owner: return f"complete or renew the saved claim in {owner}; this handoff already has an owner" + if (data.get("handoff_view") or {}).get("workflow") == "council-decision": + return f"inspect the Council evidence, then squad finish {run_id} with explicit disposition, --choose, --reason and --validation" return f'squad finish {run_id} --accept --reason="your assessment of the saved evidence"' - if action in {"continue_headless_lead", "resume_candidate_review", "handoff_submission_saved"}: + if action in {"continue_headless_lead", "resume_candidate_review", "resume_council_stage", "handoff_submission_saved"}: return f"squad resume {run_id}" if action == "recovery_file_required": return f"squad status {run_id} --json; inspect the recovery evidence before squad resume {run_id} --recovery-file FILE" @@ -680,7 +784,9 @@ def _next_lines(data: dict[str, Any]) -> list[str]: lines = [f"Next: {_next_command(data)}"] if data.get("next_action") == "claim_handoff": if (data.get("handoff_view") or {}).get("pending_finish"): - lines.append("Guidance: retry the exact saved intent; disposition and reason must match.") + lines.append("Guidance: retry the exact saved intent; all disposition, reason and choice inputs must match.") + elif (data.get("handoff_view") or {}).get("workflow") == "council-decision": + lines.append("Guidance: choose A, B or synthesis yourself; --supported-claim and --discarded-alternative repeat; dissent is retained. Extra rounds require a new capped Council run.") elif not (data.get("handoff") or {}).get("claimed_by"): lines.append("Guidance: use --reject or --revise instead of --accept if the evidence requires it.") return lines @@ -690,6 +796,21 @@ def _handoff_lines(data: dict[str, Any]) -> list[str]: view = data.get("handoff_view") if not view: return [f"Evidence unavailable: {data['handoff_view_error']}"] if data.get("handoff_view_error") else [] + if view.get("workflow") == "council-decision": + lines = ["Council: independent proposals committed before the distinct critic; automatic use off."] + for label, proposal in sorted(view["proposals"].items()): + lines.append(f"Proposal {label}: {proposal['summary']}") + critique = view["critique"] + lines.append(f"Critic: {critique['summary']}") + for assessment in critique["assessments"]: + lines.append(f"Criterion {assessment['criterion_id']} ({assessment['label']}): {assessment['status']} — {assessment['reason']}") + for objection in critique["objections"]: + lines.append(f"Dissent {objection['id']} ({objection['label']}): {objection['reason']}") + for check in view["checks"]: + lines.append(f"Check {check['id']}: {check['status']}") + artifact = view.get("report_artifact") + lines.append(f"Evidence {artifact['name']}: {artifact['path']}" if artifact else "Evidence: verified saved Council handoff report") + return lines review = view["review"] lines = [f"Review: {review['verdict']} — {review['summary']}"] for finding in review.get("findings", []): @@ -721,6 +842,8 @@ def _human_response(command: str, response: dict[str, Any]) -> str: lines.append(f" Next: {auth['next_action']}") for workflow, row in data.get("supported_workflows", {}).items(): lines.append(f"{workflow}: {'ready' if row.get('ready') else 'unavailable' if not row.get('supported') else 'needs attention'}") + if row.get("implemented_partial"): + lines.append(f" Implemented partial; native ready: false; automatic off. {row['reason']}") for row in data.get("local_apps", []): lines.append(f"{row.get('id', 'app')} registration: {row.get('status', 'unknown')}") lines.append("Next: squad setup --dry-run" if not data.get("ready") else "Next: squad review --base main --dry-run") @@ -729,7 +852,7 @@ def _human_response(command: str, response: dict[str, Any]) -> str: for row in data.get("hosts", []): lines.append(f"{row.get('id', 'app')}: {row.get('action', 'unknown')}; registration {'ready' if row.get('ready') else 'needs attention'}") lines.append("Next: squad setup" if data.get("dry_run") and data.get("completed") else "Next: squad doctor") - elif command in {"review", "fix"}: + elif command in {"review", "fix", "council"}: lines.append(f"{data.get('workflow', command)}: {data.get('state', 'unknown')}") if data.get("run_id"): lines.append(f"Run: {data['run_id']}") @@ -745,6 +868,9 @@ def _human_response(command: str, response: dict[str, Any]) -> str: lines.append(f"Check {check['id']}: {shlex.join(check['argv'])} ({'required' if check['required_to_pass'] else 'report only'})") if not data.get("check_plan"): lines.append(f"Checks: {', '.join(data.get('checks', []))}") + if command == "council": + lines.append(f"Worker cap: {data['budget']['max_worker_invocations']}; rounds: 1; lead: {data['lead']['mode']}; automatic off") + lines.append("Native readiness: unavailable until exact-boundary backend attestation; dry-run is preparation only.") next_data = data.get("service") or data lines.extend(_handoff_lines(next_data)) lines.extend(_next_lines(next_data)) diff --git a/plugin/core/src/devsquad/codex_review_worker.py b/plugin/core/src/devsquad/codex_review_worker.py index 8bece54..b94c4f7 100644 --- a/plugin/core/src/devsquad/codex_review_worker.py +++ b/plugin/core/src/devsquad/codex_review_worker.py @@ -268,10 +268,17 @@ def _isolated_codex_environment( return home, environment -def run(snapshot: dict[str, Any]) -> dict[str, Any]: +def run(snapshot: dict[str, Any], *, council_role: str | None = None, + council_prompt: str | None = None) -> dict[str, Any]: if not isinstance(snapshot, dict): raise ContractError("workflow snapshot must be an object") - adapter, profile = _validated_adapter(snapshot) + schema = review_output_schema() + if council_role is not None: + from .council import output_schema + schema = output_schema(council_role) + adapter, profile = _validated_adapter(snapshot, role=council_role or "reviewer", + adapter_key="council_adapter" if council_role else "review_adapter", + output_schema=schema) binary = Path(adapter["binary"]) try: resolved = binary.resolve(strict=True) @@ -330,6 +337,18 @@ def record(message: dict[str, Any]) -> None: try: codex_home, environment = _isolated_codex_environment(adapter["auth_file"]) + if council_role is not None: + import shutil + from .council_isolation import command + auth = Path(codex_home.name) / "auth.json" + auth.unlink() + shutil.copyfile(adapter["auth_file"], auth) + auth.chmod(0o600) + scratch = Path(codex_home.name).resolve(strict=True) + environment["CODEX_HOME"] = str(scratch) + environment["HOME"] = str(scratch) + environment["TMPDIR"] = str(scratch) + argv = command(snapshot["council_boundary"], argv, scratch=scratch) process = subprocess.Popen( argv, cwd=review_root, @@ -409,7 +428,7 @@ def record(message: dict[str, Any]) -> None: or Path(reported_cwd).resolve() != review_root): raise ContractError("Codex thread did not preserve the frozen execution identity") - prompt = build_review_prompt(snapshot["task"], workspace) + prompt = council_prompt if council_role is not None else build_review_prompt(snapshot["task"], workspace) peer.send(turn_start_request( 101, thread_id=thread_id, @@ -418,7 +437,7 @@ def record(message: dict[str, Any]) -> None: effort=effort, cwd=str(review_root), permission="read_only", - output_schema=review_output_schema(), + output_schema=schema, )) turn_result = _response_result( receive_response( @@ -463,9 +482,11 @@ def record(message: dict[str, Any]) -> None: + canonical_json(state.output_diagnostics()) ) try: - review = decode_review_document( - review_payload, snapshot["task"], workspace, - ) + if council_role is not None: + from .council import decode + review = decode(review_payload) + else: + review = decode_review_document(review_payload, snapshot["task"], workspace) except ContractError as exc: raise ContractError( f"{exc}; protocol_summary=" @@ -487,6 +508,9 @@ def record(message: dict[str, Any]) -> None: "permission_policy": "read_only", "verification": "verified", } + if council_role is not None: + return {"document": review, "observed_identity": observed_identity, + "native_ids": {"thread_id": thread_id, "turn_id": turn_id}, "usage": usage} return run_review_and_checks( snapshot, review, diff --git a/plugin/core/src/devsquad/contracts.py b/plugin/core/src/devsquad/contracts.py index 0f8a820..9c0d1b2 100644 --- a/plugin/core/src/devsquad/contracts.py +++ b/plugin/core/src/devsquad/contracts.py @@ -94,9 +94,11 @@ def __post_init__(self) -> None: raise ContractError("stdin_path must be a non-empty string or null") if not isinstance(self.requested, ExecutionIdentity): raise ContractError("requested must be an execution identity") - allowed_env = {"DEVSQUAD_WORKER", "DEVSQUAD_RUN_ID", "DEVSQUAD_ATTEMPT_ID", "DEVSQUAD_DELEGATION_DEPTH"} + allowed_env = {"DEVSQUAD_WORKER", "DEVSQUAD_RUN_ID", "DEVSQUAD_ATTEMPT_ID", "DEVSQUAD_DELEGATION_DEPTH", "DEVSQUAD_COUNCIL_ROLE"} if not isinstance(self.environment, dict) or set(self.environment) - allowed_env or not all(isinstance(k, str) and isinstance(v, str) for k, v in self.environment.items()): raise ContractError("environment contains non-allowlisted or non-string values") + if self.environment.get("DEVSQUAD_COUNCIL_ROLE", "") not in {"", "proposer_a", "proposer_b", "critic", "lead"}: + raise ContractError("Council worker role marker is invalid") def to_dict(self) -> dict[str, Any]: return asdict(self) diff --git a/plugin/core/src/devsquad/council.py b/plugin/core/src/devsquad/council.py new file mode 100644 index 0000000..6055604 --- /dev/null +++ b/plugin/core/src/devsquad/council.py @@ -0,0 +1,242 @@ +"""Strict contracts for the explicitly invoked, read-only Council workflow.""" +from __future__ import annotations + +import hashlib +import json +from typing import Any + +from .contracts import ContractError +from .store import canonical_json + +ROLES = ("proposer_a", "proposer_b", "critic") +MAX_PACKET_BYTES = 512 * 1024 +PROMPT_VERSION = "council-role-packet-v1" + + +def role_prompt(role: str, packet: dict) -> str: + return (f"You are the single read-only Council {role}. Read evidence.json in your directory. " + "Use the frozen rubric and cite only supplied evidence IDs. Produce the requested structured document. " + "No peer research or implementation writes. The lead retains every objection ID and required validation. " + "Preference cannot override a failed mandatory check.\n" + canonical_json(packet)) + + +def prompt_digest(role: str, packet: dict) -> str: + return hashlib.sha256(role_prompt(role, packet).encode()).hexdigest() + + +def digest(value: Any) -> str: + return hashlib.sha256(canonical_json(value).encode()).hexdigest() + + +def exact(value: Any, fields: set[str], label: str) -> dict: + if not isinstance(value, dict) or set(value) != fields: + raise ContractError(f"{label} fields are invalid") + return value + + +def text(value: Any, label: str) -> str: + if not isinstance(value, str) or not value.strip() or len(value) > 16000: + raise ContractError(f"{label} must be bounded non-empty text") + return value + + +def sha(value: Any) -> str: + if not isinstance(value, str) or len(value) != 64 or any(c not in "0123456789abcdef" for c in value): + raise ContractError("Council hash is invalid") + return value + + +def validate_spec(spec: Any) -> dict: + exact(spec, {"schema_version", "enabled", "automatic", "reason", "min_valid_proposals", + "required_critics", "max_invocations", "seed", "evidence", "rubric"}, "CouncilSpec") + if type(spec["schema_version"]) is not int or spec["schema_version"] != 1: + raise ContractError("CouncilSpec version is invalid") + if spec["enabled"] is not True or spec["automatic"] is not False: + raise ContractError("Council requires explicit invocation; automatic Council is disabled") + if type(spec["min_valid_proposals"]) is not int or spec["min_valid_proposals"] != 2: + raise ContractError("Council requires two valid proposals") + if type(spec["required_critics"]) is not int or spec["required_critics"] != 1: + raise ContractError("Council requires one distinct critic") + if type(spec["max_invocations"]) is not int or not 3 <= spec["max_invocations"] <= 16: + raise ContractError("Council invocation cap must be between 3 and 16") + text(spec["reason"], "Council reason") + sha(spec["seed"]) + if not isinstance(spec["evidence"], list) or len(spec["evidence"]) > 32: + raise ContractError("Council evidence must be a bounded array") + identifiers = set() + for item in spec["evidence"]: + exact(item, {"artifact_id", "sha256"}, "Council evidence reference") + text(item["artifact_id"], "artifact_id") + sha(item["sha256"]) + if item["artifact_id"] in identifiers: + raise ContractError("Council evidence IDs must be unique") + identifiers.add(item["artifact_id"]) + if not isinstance(spec["rubric"], list) or not 1 <= len(spec["rubric"]) <= 32: + raise ContractError("Council rubric must be a non-empty bounded array") + identifiers = set() + for item in spec["rubric"]: + exact(item, {"id", "description"}, "Council rubric item") + text(item["id"], "rubric id") + text(item["description"], "rubric description") + if item["id"] in identifiers: + raise ContractError("Council rubric IDs must be unique") + identifiers.add(item["id"]) + return spec + + +def label_mapping(seed: str, authors: list[str]) -> dict[str, str]: + sha(seed) + if len(authors) != 2 or len(set(authors)) != 2: + raise ContractError("Council label mapping requires two distinct authors") + ordered = sorted(authors, key=lambda author: hashlib.sha256((seed + ":" + author).encode()).hexdigest()) + return dict(zip(("A", "B"), ordered)) + + +def evidence_ids(value: Any, spec: dict) -> list[str]: + allowed = {ref["artifact_id"] for ref in spec["evidence"]} | set(spec.get("source_ids", [])) + if (not isinstance(value, list) or not all(isinstance(v, str) and v in allowed for v in value) + or len(value) != len(set(value))): + raise ContractError("Council evidence IDs are unknown or duplicated") + return value + + +def sanitized_proposals(snapshot: dict) -> dict: + """Strip supplied identity tokens from text; raw originals remain sealed.""" + import re + mapping = label_mapping(snapshot["task"]["council"]["seed"], ["proposer_a", "proposer_b"]) + documents = snapshot["council_state"]["documents"] + sensitive = set() + for role in ("proposer_a", "proposer_b"): + evidence = documents[role] + profile = evidence["profile"] + sensitive.update([role, profile["id"], profile["model_id"], profile["account_pool_id"]]) + sensitive.update(str(value) for value in (evidence.get("observed_identity") or {}).values() + if isinstance(value, str) and len(value) > 3) + def sanitize(value, key=None): + if key == "evidence_ids": + return value # Frozen provenance IDs must not be rewritten. + if isinstance(value, str): + for token in sorted(sensitive, key=len, reverse=True): + value = re.sub(re.escape(token), "[identity omitted]", value, flags=re.IGNORECASE) + return value + if isinstance(value, list): + return [sanitize(item) for item in value] + if isinstance(value, dict): + return {field: sanitize(item, field) for field, item in value.items()} + return value + return {label: sanitize(documents[author]["document"]) for label, author in mapping.items()} + + +def validate_proposal(value: Any, spec: dict) -> dict: + exact(value, {"summary", "approach", "claims", "validation"}, "Council proposal") + for field in ("summary", "approach", "validation"): + text(value[field], field) + if not isinstance(value["claims"], list) or not 1 <= len(value["claims"]) <= 64: + raise ContractError("Council proposal claims must be non-empty and bounded") + for claim in value["claims"]: + exact(claim, {"text", "evidence_ids"}, "Council claim") + text(claim["text"], "claim text") + evidence_ids(claim["evidence_ids"], spec) + return value + + +def validate_critique(value: Any, spec: dict) -> dict: + exact(value, {"summary", "assessments", "objections"}, "Council critique") + text(value["summary"], "critique summary") + required = {(label, criterion["id"]) for label in ("A", "B") for criterion in spec["rubric"]} + seen = set() + if not isinstance(value["assessments"], list) or len(value["assessments"]) != len(required): + raise ContractError("Council critique must assess every label and rubric criterion") + for item in value["assessments"]: + exact(item, {"label", "criterion_id", "status", "reason", "evidence_ids"}, "Council assessment") + pair = (item["label"], item["criterion_id"]) + if pair not in required or pair in seen or item["status"] not in {"supported", "unsupported", "uncertain"}: + raise ContractError("Council assessment label/criterion/status is invalid or duplicated") + seen.add(pair) + text(item["reason"], "assessment reason") + evidence_ids(item["evidence_ids"], spec) + if not isinstance(value["objections"], list) or len(value["objections"]) > 64: + raise ContractError("Council objections must be bounded") + seen = set() + for item in value["objections"]: + exact(item, {"id", "label", "reason", "evidence_ids"}, "Council objection") + if item["label"] not in {"A", "B"} or item["id"] in seen: + raise ContractError("Council objection label/ID is invalid") + seen.add(text(item["id"], "objection id")) + text(item["reason"], "objection reason") + evidence_ids(item["evidence_ids"], spec) + return value + + +def validate_choice(value: Any, spec: dict, critique: dict) -> dict: + exact(value, {"disposition", "reason", "chosen", "supported_claims", "discarded_alternatives", + "unresolved_objections", "validation"}, "Council decision") + if value["disposition"] not in {"accept", "reject", "revise"} or value["chosen"] not in {"A", "B", "synthesis"}: + raise ContractError("Council disposition/chosen label is invalid") + for field in ("reason", "validation"): + text(value[field], field) + for field in ("supported_claims", "discarded_alternatives", "unresolved_objections"): + if not isinstance(value[field], list) or len(value[field]) > 64 or not all(isinstance(v, str) and v for v in value[field]): + raise ContractError(f"Council decision {field} must be a bounded text array") + for entry in value[field]: + text(entry, f"Council decision {field} entry") + objections = {item["id"] for item in critique["objections"]} + # Retain every objection; disposition does not silently erase a dissenting source. + if set(value["unresolved_objections"]) != objections or len(value["unresolved_objections"]) != len(objections): + raise ContractError("Council decision must retain every recorded objection ID") + if value["disposition"] == "accept" and not value["supported_claims"]: + raise ContractError("Council acceptance needs supported claims") + return value + + +def output_schema(role: str) -> dict: + string = {"type": "string", "minLength": 1} + array = {"type": "array", "items": string} + def obj(properties): + return {"type": "object", "additionalProperties": False, "required": list(properties), "properties": properties} + if role in {"proposer_a", "proposer_b"}: + return obj({"summary": string, "approach": string, "claims": {"type": "array", "minItems": 1, + "items": obj({"text": string, "evidence_ids": array})}, "validation": string}) + if role == "critic": + return obj({"summary": string, "assessments": {"type": "array", "items": obj({ + "label": {"enum": ["A", "B"]}, "criterion_id": string, + "status": {"enum": ["supported", "unsupported", "uncertain"]}, "reason": string, + "evidence_ids": array})}, "objections": {"type": "array", "items": obj({ + "id": string, "label": {"enum": ["A", "B"]}, "reason": string, "evidence_ids": array})}}) + if role == "lead": + return obj({"disposition": {"enum": ["accept", "reject", "revise"]}, "reason": string, + "chosen": {"enum": ["A", "B", "synthesis"]}, "supported_claims": array, + "discarded_alternatives": array, "unresolved_objections": array, "validation": string}) + raise ContractError("Council role is invalid") + + +def verify_identities(documents: dict[str, dict], *, fixture: bool) -> None: + identities = [] + native_ids = set() + for role in ROLES: + value = documents.get(role) + if not isinstance(value, dict): + raise ContractError(f"Council is missing {role}; no valid quorum") + identity = value.get("observed_identity") + if fixture: + if identity is not None or value.get("identity_scope") != "all_fixture": + raise ContractError("Council mixed fixture/native identities cannot establish quorum") + identities.append(value["profile"]["model_id"].casefold()) + else: + if (not isinstance(identity, dict) or identity.get("verification") != "verified" + or not isinstance(identity.get("model_id"), str) or not identity["model_id"]): + raise ContractError("Council requires verified observed model identities") + identities.append(identity["model_id"].casefold()) + ids = value.get("native_ids", {}) + pair = (ids.get("thread_id"), ids.get("turn_id")) + if not all(isinstance(item, str) and item for item in pair) or pair in native_ids: + raise ContractError("Council native role turns must be distinct correlated thread/turn identities") + native_ids.add(pair) + if len(set(identities)) != 3: + raise ContractError("Council critic/proposers must have three distinct model identities") + + +def decode(value: bytes | str) -> dict: + from .workflows import _strict_json_object + result = _strict_json_object(value, "Council evidence", maximum=MAX_PACKET_BYTES) + return result diff --git a/plugin/core/src/devsquad/council_comparison.py b/plugin/core/src/devsquad/council_comparison.py new file mode 100644 index 0000000..b418118 --- /dev/null +++ b/plugin/core/src/devsquad/council_comparison.py @@ -0,0 +1,166 @@ +"""Predeclared workflow mechanics comparison, separate from R3 eligibility. + +Public fixture runs support observable overhead/failure mechanics, not native +quality. No fabricated score, profile promotion or automatic Council authority +can be produced by this bounded comparison. +""" +from __future__ import annotations + +from datetime import datetime, timezone +import hashlib +import json +import os +from pathlib import Path + +from .contracts import ContractError +from .council import digest, exact, sha, text +from .store import canonical_json, ConflictError + +ARMS = ("control", "council") +CONTRACT_FIELDS = ("project", "goal", "acceptance", "checks", "scope") + + +def _file_sha(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _task_version(task: dict) -> dict: + if set(task["routing"]) != {"profiles", "policy"}: + raise ContractError("Workflow comparison requires embedded frozen routing documents") + module = "council.py" if task["workflow"] == "council-decision" else "workflows.py" + return {"task_sha256": digest(task), "profiles_sha256": digest(task["routing"]["profiles"]), + "policy_sha256": digest(task["routing"]["policy"]), "prompt_module": module, + "prompt_module_sha256": _file_sha(Path(__file__).with_name(module))} + + +def predeclare(path: Path, cases: list[dict]) -> dict: + """Create once before runs. Each case fixes both task versions and split.""" + if not isinstance(cases, list) or not 2 <= len(cases) <= 16: + raise ContractError("Comparison requires bounded matched and held-out cases") + declared, identifiers, splits, contracts = [], set(), set(), set() + for case in cases: + exact(case, {"id", "split", "control", "council"}, "workflow comparison case") + text(case["id"], "comparison case ID") + if case["id"] in identifiers or case["split"] not in {"matched", "heldout"}: + raise ContractError("Comparison case IDs/splits are invalid") + identifiers.add(case["id"]) + splits.add(case["split"]) + control, council = case["control"], case["council"] + if control["workflow"] != "branch-review" or council["workflow"] != "council-decision": + raise ContractError("Comparison arms must be branch-review and Council") + contract = {field: control[field] for field in CONTRACT_FIELDS} + if contract != {field: council[field] for field in CONTRACT_FIELDS}: + raise ContractError("Comparison arms do not share the declared input/check contract") + contract_sha256 = digest(contract) + if contract_sha256 in contracts: + raise ContractError("Comparison cases must have unique input contracts; repetitions are not held-out independence") + contracts.add(contract_sha256) + for task in (control, council): + for field in ("base_ref", "target_ref"): + value = task["project"][field] + if not isinstance(value, str) or len(value) not in {40, 64} or any(c not in "0123456789abcdef" for c in value): + raise ContractError("Comparison must pin exact Git commit IDs") + declared.append({"id": case["id"], "split": case["split"], "input_contract_sha256": contract_sha256, + **{arm: _task_version(case[arm]) for arm in ARMS}}) + if splits != {"matched", "heldout"}: + raise ContractError("Comparison must predeclare both matched and held-out cases") + plan = {"schema_version": 1, "created_at": datetime.now(timezone.utc).isoformat(), + "scope": "public_fixture_mechanics", "automatic_enabled": False, + "quality_gate": {"accepted_quality": "unmeasured", "escaped_defects": "unmeasured", "rework": "unmeasured"}, + "cases": declared, "decision_rule": "inconclusive_without_independent_native_quality_evidence"} + path.parent.mkdir(parents=True, exist_ok=True) + descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0), 0o600) + with os.fdopen(descriptor, "wb") as stream: + stream.write(canonical_json(plan).encode()) + stream.flush() + os.fsync(stream.fileno()) + return {"path": str(path), "sha256": _file_sha(path)} + + +def report(store, declaration: dict, pairs: list[dict]) -> dict: + """Validate actual terminal public receipts, then save an immutable report.""" + exact(declaration, {"path", "sha256"}, "comparison declaration reference") + sha(declaration["sha256"]) + path = Path(declaration["path"]) + if path.is_symlink() or _file_sha(path) != declaration["sha256"]: + raise ConflictError("Workflow comparison predeclaration changed") + plan = json.loads(path.read_bytes()) + if plan.get("scope") != "public_fixture_mechanics" or plan.get("automatic_enabled") is not False: + raise ContractError("Comparison declaration is not the bounded public fixture gate") + if len({case["input_contract_sha256"] for case in plan["cases"]}) != len(plan["cases"]): + raise ContractError("Comparison repetitions cannot be relabelled independent matched/held-out cases") + if not isinstance(pairs, list) or len(pairs) != len(plan["cases"]): + raise ContractError("Comparison requires every predeclared case") + assigned = {item["id"]: item for item in pairs} + if len(assigned) != len(pairs) or set(assigned) != {item["id"] for item in plan["cases"]}: + raise ContractError("Comparison pairs differ from the predeclared cases") + created = datetime.fromisoformat(plan["created_at"]) + results, seen = [], set() + for case in plan["cases"]: + pair = assigned[case["id"]] + exact(pair, {"id", "control", "council"}, "comparison pair") + arms = {} + for arm in ARMS: + run_id = pair[arm] + if run_id in seen: + raise ContractError("A comparison run cannot be reused across arms/cases") + seen.add(run_id) + run = store.run(run_id) + if datetime.fromisoformat(run["created_at"]) <= created or run["state"] not in {"succeeded", "failed", "cancelled"}: + raise ConflictError("Comparison run was not terminal after its predeclaration") + snapshot = json.loads(run["mutable_snapshot"]) + task = snapshot["task"] + requested = json.loads(run["submitted_request"])["task"] + if digest(requested) != case[arm]["task_sha256"]: + raise ConflictError("Comparison run task differs from its declared arm") + if digest({field: task[field] for field in CONTRACT_FIELDS}) != case["input_contract_sha256"]: + raise ConflictError("Comparison actual frozen input contract differs") + if (snapshot["routing"]["profile_registry"]["sha256"] != case[arm]["profiles_sha256"] + or snapshot["routing"]["policy"]["sha256"] != case[arm]["policy_sha256"] + or _file_sha(Path(run["package_path"]) / "devsquad" / case[arm]["prompt_module"]) != case[arm]["prompt_module_sha256"]): + raise ConflictError("Comparison frozen profile/policy/prompt version differs") + if (arm == "council" and snapshot.get("council_fixture") is None + or arm == "control" and "internal_review_fixture" not in snapshot): + raise ContractError("This comparison gate accepts actual public fixture mechanics only") + artifact = store.artifact_named(run_id, "receipt.json") + if artifact is None: + raise ConflictError("Comparison run has no terminal workflow receipt") + from .council_runtime import verified_artifact + raw = verified_artifact(store, run_id, "receipt.json", artifact["sha256"]) + receipt = json.loads(raw) + if receipt["run_id"] != run_id or receipt["state"] != run["state"]: + raise ConflictError("Comparison receipt identity/state differs") + attempts = receipt["attempts"] + accounting = receipt.get("accounting", receipt) + arms[arm] = {"run_id": run_id, "receipt": {"artifact_id": artifact["id"], "sha256": artifact["sha256"]}, + "state": run["state"], "lead_disposition": receipt["lead"]["disposition"], + "runtime_package_sha256": run["package_digest"], + "input_contract_sha256": case["input_contract_sha256"], "versions": case[arm], + "actual_prompt_sha256": [item.get("prompt_sha256") for item in attempts], + "candidate": receipt["candidate"], "brief_sha256": receipt.get("brief_sha256"), + "accepted_quality": None, "escaped_defects": None, "rework": None, + "quality_missingness": "fixture mechanics and host acceptance are not independent native quality measurements", + "execution_elapsed_ms": store._execution_elapsed_ms(run_id, run, datetime.now(timezone.utc)), + "worker_invocations": accounting["worker_invocations"], + "native_usage": [item.get("usage") for item in attempts], + "native_model_requests": accounting.get("native_model_requests"), "native_quota": None, + "host_usage": None, "identity_scope": "all_fixture"} + results.append({"id": case["id"], "split": case["split"], "arms": arms}) + value = {"schema_version": 1, "predeclaration_sha256": declaration["sha256"], "cases": results, + "conclusion": "inconclusive", "quality_benefit_supported": False, "automatic_enabled": False, + "limitations": ["Controlled process mechanics only", "No independent native accepted-quality/escaped-defect/rework observations", + "Workflow prompts and invocation count differ; this is not R3 single-binding eligibility"]} + # Terminal workflow ledgers stay immutable. This separately versioned + # comparison artifact references their exact receipts, never rewrites them. + destination = path.parent / (declaration["sha256"] + ".workflow-comparison.json") + encoded = canonical_json(value).encode() + if destination.exists(): + if destination.is_symlink() or destination.read_bytes() != encoded: + raise ConflictError("Workflow comparison report is already frozen differently") + else: + descriptor = os.open(destination, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0), 0o600) + with os.fdopen(descriptor, "wb") as stream: + stream.write(encoded) + stream.flush() + os.fsync(stream.fileno()) + return {"report": value, "path": str(destination), "sha256": digest(value)} diff --git a/plugin/core/src/devsquad/council_isolation.py b/plugin/core/src/devsquad/council_isolation.py new file mode 100644 index 0000000..cf01de9 --- /dev/null +++ b/plugin/core/src/devsquad/council_isolation.py @@ -0,0 +1,141 @@ +"""Versioned macOS default-deny read boundary for untrusted Council participants.""" +from __future__ import annotations + +import hashlib +import json +from pathlib import Path +import platform +import subprocess +import os + +from .contracts import CapabilityUnavailable, ContractError + +BOUNDARY_VERSION = 1 + + +def _literal(path: Path) -> str: + # JSON escaping safely quotes Seatbelt strings without shell interpolation. + return json.dumps(str(path.resolve(strict=True))) + + +def profile(*, executable: Path, evidence: Path, scratch: Path | None = None, + dependencies: tuple[Path, ...] = ()) -> str: + executable = executable.resolve(strict=True) + evidence = evidence.resolve(strict=True) + roots = [evidence, *[p.resolve(strict=True) for p in dependencies]] + system_roots = [Path("/usr/lib"), Path("/System/Library")] + ancestors = {Path("/")} + for root in [executable, *roots, *system_roots, *([scratch.resolve(strict=True)] if scratch else [])]: + ancestors.update(root.parents) + literals = " ".join(f"(literal {_literal(p)})" for p in sorted(ancestors, key=str)) + read_roots = " ".join(f"(subpath {_literal(p)})" for p in [*system_roots, *roots]) + result = ("(version 1)\n(deny default)\n(allow process-fork)\n(allow sysctl-read)\n" + f"(allow process-exec (literal {_literal(executable)}))\n" + f"(allow file-read* (literal {_literal(executable)}) {literals} {read_roots})\n") + if scratch is not None: + result += f"(allow file-read* file-write* (subpath {_literal(scratch)}))\n" + return result + + +def freeze_boundary(*, executable: Path, evidence: Path, dependencies: tuple[Path, ...] = (), + native_codex: bool = False) -> dict: + if platform.system() != "Darwin" or not Path("/usr/bin/sandbox-exec").is_file(): + raise CapabilityUnavailable("Council requires the verified macOS Seatbelt read boundary") + value = profile(executable=executable, evidence=evidence, dependencies=dependencies) + if native_codex: + # Codex probes absent managed config paths during bootstrap. Metadata + # only, not managed-config bytes/MCP configuration, may be inspected. + value += ('(allow file-read-metadata (literal "/etc") (literal "/private/etc") ' + '(subpath "/etc/codex") (subpath "/private/etc/codex"))\n' + '(allow file-read* (literal "/dev/urandom"))\n' + '(allow mach-lookup (global-name "com.apple.cfprefsd.daemon") (global-name "com.apple.cfprefsd.agent"))\n' + f'(allow ipc-posix-shm-read-data (ipc-posix-name "apple.cfprefs.{os.getuid()}v1") (ipc-posix-name "apple.cfprefs.daemonv1"))\n' + '(allow network-outbound (remote tcp "*:443"))\n') + return {"schema_version": BOUNDARY_VERSION, "platform": platform.system(), + "sandbox_binary": "/usr/bin/sandbox-exec", "profile": value, + "profile_sha256": hashlib.sha256(value.encode()).hexdigest(), + "executable_sha256": hashlib.sha256(executable.resolve(strict=True).read_bytes()).hexdigest(), + "dependency_roots": [str(p.resolve(strict=True)) for p in dependencies], + "native_codex": native_codex, "system_read_roots": ["/usr/lib", "/System/Library"], + "uid": os.getuid(), + "network": "outbound_tcp_443" if native_codex else "denied"} + + +def command(boundary: dict, argv: list[str], *, scratch: Path | None = None) -> list[str]: + if (boundary.get("schema_version") != BOUNDARY_VERSION or boundary.get("platform") != "Darwin" + or boundary.get("uid") != os.getuid() + or platform.system() != "Darwin" or boundary.get("sandbox_binary") != "/usr/bin/sandbox-exec" + or hashlib.sha256(boundary.get("profile", "").encode()).hexdigest() != boundary.get("profile_sha256") + or hashlib.sha256(Path(argv[0]).read_bytes()).hexdigest() != boundary.get("executable_sha256")): + raise CapabilityUnavailable("frozen Council read boundary changed or is unavailable") + value = boundary["profile"] + if scratch is not None: + # Provider-owned auth/session cache; never contains peer artifacts or the ledger. + value += f"(allow file-read* file-write* (subpath {_literal(scratch)}))\n" + for ancestor in scratch.resolve(strict=True).parents: + value += f"(allow file-read* (literal {_literal(ancestor)}))\n" + return [boundary["sandbox_binary"], "-p", value, *argv] + + +def probe(boundary: dict, *, executable: Path, own_file: Path, forbidden: tuple[Path, ...]) -> None: + for path, allowed in [(own_file, True), *[(p, False) for p in forbidden]]: + result = subprocess.run(command(boundary, [str(executable), str(path)]), + capture_output=True, timeout=3, check=False) + if (allowed and result.returncode != 0) or (not allowed and result.returncode == 0): + raise CapabilityUnavailable("Council filesystem read isolation probe failed") + + +def verify_read_boundary(boundary: dict, *, own_file: Path, forbidden: tuple[Path, ...]) -> None: + """Probe the frozen file rules; the diagnostic cat executable alone is added.""" + value = boundary["profile"] + '(allow process-exec (literal "/bin/cat"))\n(allow file-read* (literal "/bin/cat") (literal "/bin"))\n' + for path, allowed in [(own_file, True), *[(p, False) for p in forbidden]]: + result = subprocess.run([boundary["sandbox_binary"], "-p", value, "/bin/cat", str(path.resolve(strict=True))], + capture_output=True, timeout=3, check=False) + if (allowed and result.returncode != 0) or (not allowed and result.returncode == 0): + raise CapabilityUnavailable("Council default-deny peer/runtime read probe failed") + + +def verify_native_bootstrap(boundary: dict, *, executable: Path, expected_version: str, evidence: Path) -> None: + """Non-generating app-server bootstrap under the exact frozen boundary.""" + import tempfile + from .codex_protocol import JsonLinePeer, initialize_request, receive_response + from .diagnostics import _probe_output + from .probe_process import capture_probe_identity, close_probe + with tempfile.TemporaryDirectory(prefix="devsquad-council-bootstrap-") as temporary: + scratch = Path(temporary).resolve(strict=True) + environment = {"PATH": "/usr/bin:/bin", "CODEX_HOME": str(scratch), "HOME": str(scratch), "TMPDIR": str(scratch)} + code, output = _probe_output(command(boundary, [str(executable), "--version"], scratch=scratch), + project=evidence, environment=environment) + if code != 0 or output.strip() != expected_version: + raise CapabilityUnavailable("Council native binary cannot start inside frozen read isolation") + process = subprocess.Popen(command(boundary, [str(executable), "--disable", "apps", "--disable", "plugins", + "app-server", "--listen", "stdio://"], scratch=scratch), cwd=evidence, env=environment, + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, start_new_session=True) + start_identity = capture_probe_identity(process) + try: + if start_identity is None: + raise CapabilityUnavailable("Council native bootstrap ownership identity is unavailable") + peer = JsonLinePeer(process.stdout, process.stdin, max_frame_bytes=16 * 1024) + peer.send(initialize_request(1)) + response = receive_response(peer, 1, timeout_seconds=3) + if "error" in response or not isinstance(response.get("result"), dict): + raise CapabilityUnavailable("Council native isolated bootstrap is unavailable") + except (OSError, EOFError, TimeoutError) as exc: + raise CapabilityUnavailable("Council native isolated bootstrap is unavailable") from exc + finally: + # The shared bounded helper is the sole cleanup authority. Failure + # to prove ownership or absence remains unavailable, never ready. + close_probe(process, start_identity=start_identity) + + +def verify_native_network(boundary: dict) -> None: + """Fail before generation until exact-boundary backend connectivity is proven. + + Initialization and model/list can use local/fallback catalog data. They are + deliberately not a network attestation. No flag or unsafe launch fallback + can override this currently unavailable native capability. + """ + raise CapabilityUnavailable( + "native Council is unavailable: a genuine non-generating HTTPS backend response " + "under the exact frozen sandbox boundary has not been verified" + ) diff --git a/plugin/core/src/devsquad/council_runtime.py b/plugin/core/src/devsquad/council_runtime.py new file mode 100644 index 0000000..24aee86 --- /dev/null +++ b/plugin/core/src/devsquad/council_runtime.py @@ -0,0 +1,452 @@ +"""Council stage projections over the existing durable runner and handoff store.""" +from __future__ import annotations + +from datetime import datetime, timezone +import hashlib +import json +from pathlib import Path +from typing import Any + +from .contracts import CapabilityUnavailable, ContractError +from .council import (ROLES, MAX_PACKET_BYTES, decode, digest, label_mapping, output_schema, + sanitized_proposals, validate_choice, validate_critique, validate_proposal, verify_identities) +from .store import ConflictError, canonical_json, request_hash + + +def frozen_fields(snapshot: dict) -> dict: + return {key: snapshot[key] for key in ("task", "base_oid", "target_oid", "configs", "routing", + "workspace", "check_workspace", "council_brief", "council_directories", "council_adapters", + "council_boundaries", "council_fixture")} + + +def verify_origin(store, run_id: str, snapshot: dict) -> None: + artifact = store.artifact_named(run_id, "council-origin.json") + if (artifact is None or hashlib.sha256(Path(artifact["path"]).read_bytes()).hexdigest() != artifact["sha256"] + or digest(frozen_fields(snapshot)) != snapshot.get("council_origin_sha256") + or artifact["sha256"] != snapshot["council_origin_sha256"]): + raise ConflictError("Council frozen inputs changed or are missing") + + +def verified_artifact(store, run_id: str, name: str, expected_sha256: str | None = None) -> bytes: + artifact = store.artifact_named(run_id, name) + if artifact is None: + raise ConflictError("Council evidence artifact is missing") + path = Path(artifact["path"]) + expected_parent = (store.artifacts / run_id).resolve() + if path.is_symlink() or path.resolve(strict=True).parent != expected_parent: + raise ConflictError("Council evidence artifact escaped its private directory") + raw = path.read_bytes() + if (len(raw) != artifact["byte_size"] or hashlib.sha256(raw).hexdigest() != artifact["sha256"] + or expected_sha256 is not None and artifact["sha256"] != expected_sha256): + raise ConflictError("Council evidence artifact is corrupt") + return raw + + +def validate_saved_handoff(store, run_id: str, handoff, snapshot: dict) -> None: + """Re-derive quorum/check truth from exact imported stage artifacts.""" + verify_origin(store, run_id, snapshot) + state = snapshot["council_state"] + if set(state["documents"]) != set(ROLES) or state["next_role"] != "lead": + raise ConflictError("Council handoff has no complete sealed proposal/critic quorum") + attempts = {a["id"]: a for a in store.attempts_for_run(run_id)} + for role in ROLES: + reference = state["artifacts"][role] + evidence = decode(verified_artifact(store, run_id, reference["name"], reference["sha256"])) + matching = [a for a in attempts.values() if reference["name"] == f"council-{role}-{a['id']}.json"] + if len(matching) != 1 or matching[0]["status"] != "finished" or evidence != state["documents"][role]: + raise ConflictError("Council finalized stage projection differs from its actual imported attempt") + validate_document(evidence, snapshot, matching[0]) + verify_identities(state["documents"], fixture=snapshot["council_fixture"] is not None) + mapping = label_mapping(snapshot["task"]["council"]["seed"], ["proposer_a", "proposer_b"]) + checks = state["documents"]["critic"]["checks"] + failures = [c["id"] for c in checks if c["required_to_pass"] and c["status"] != "passed"] + integrity = [c["id"] for c in checks if c.get("integrity", {}).get("status") in {"violated", "not_run"}] + packet = handoff.packet + if (digest(packet) != handoff.packet_sha256 or packet.get("workflow") != "council-decision" + or packet.get("brief_sha256") != digest(snapshot["council_brief"]) + or packet.get("candidate_sha256") != snapshot["workspace"]["candidate_sha256"] + or packet.get("base_oid") != snapshot["base_oid"] or packet.get("target_oid") != snapshot["target_oid"] + or packet.get("label_mapping") != mapping or packet.get("checks") != checks + or packet.get("critique") != state["documents"]["critic"]["document"] + or packet.get("proposals") != sanitized_proposals(snapshot) + or packet.get("evaluation") != {"accept_allowed": not failures and not integrity, + "required_failures": failures, "integrity_failures": integrity}): + raise ConflictError("Council saved handoff differs from frozen evidence/check truth") + for reference in packet["artifacts"]: + raw = verified_artifact(store, run_id, reference["name"], reference["sha256"]) + if reference["name"].startswith("council-evidence-") and decode(raw) != { + "documents": state["documents"], "label_mapping": mapping, "brief_sha256": digest(snapshot["council_brief"])}: + raise ConflictError("Council raw provenance bundle differs from finalized evidence") + + +def prepare(snapshot: dict, *, store, run_id: str, runtime: Path, fixture: dict | None) -> None: + from .codex_review_worker import freeze_codex_role + from .council_isolation import freeze_boundary, verify_read_boundary, verify_native_bootstrap + from .workspaces import committed_regular_file, _git + + task = snapshot["task"] + if fixture is not None and not isinstance(fixture, dict): + raise ContractError("Council fixture must be an object") + brief = {"schema_version": 1, "goal": task["goal"], "base_oid": snapshot["base_oid"], + "target_oid": snapshot["target_oid"], "scope": task["scope"], "acceptance": task["acceptance"], + "rubric": task["council"]["rubric"], "evidence": [], "source_files": []} + for ref in task["council"]["evidence"]: + row = store.connection.execute( + "SELECT a.*,r.project_id FROM artifacts a JOIN runs r ON r.id=a.run_id WHERE a.id=?", (ref["artifact_id"],) + ).fetchone() + if row is None or row["project_id"] != store.run(run_id)["project_id"] or row["sha256"] != ref["sha256"]: + raise ContractError("Council evidence is missing, belongs to another project, or has changed") + raw = Path(row["path"]).read_bytes() + if len(raw) != row["byte_size"] or hashlib.sha256(raw).hexdigest() != ref["sha256"]: + raise ConflictError("Council evidence artifact is corrupt") + try: + content = raw.decode("utf-8") + except UnicodeDecodeError as exc: + raise ContractError("Council evidence must be UTF-8 text") from exc + brief["evidence"].append({**ref, "content": content}) + repo = Path(task["project"]["repo_path"]) + files = _git(repo, "ls-tree", "-r", "--name-only", "-z", snapshot["target_oid"]).split(b"\0") + for encoded in files: + if not encoded: + continue + path = encoded.decode("utf-8", "surrogateescape") + if not any(scope == "." or path == scope or path.startswith(scope.rstrip("/") + "/") for scope in task["scope"]["read_paths"]): + continue + try: + raw = committed_regular_file(repo, snapshot["target_oid"], path) + content = raw.decode("utf-8") + except UnicodeDecodeError as exc: + raise ContractError("Council scoped source must be UTF-8; narrow the read scope") from exc + source_hash = hashlib.sha256(raw).hexdigest() + brief["source_files"].append({"path": path, "artifact_id": f"source:{path}:{source_hash}", "sha256": source_hash, "content": content}) + if len(canonical_json(brief).encode()) > MAX_PACKET_BYTES: + raise ContractError("Council evidence exceeds its cap; narrow the read scope") + if len(canonical_json(brief).encode()) > MAX_PACKET_BYTES: + raise ContractError("Council evidence exceeds its byte cap") + snapshot.update(council_brief=brief, council_directories={}, council_adapters={}, council_boundaries={}, + council_fixture=fixture, council_state={"documents": {}, "next_role": "proposer_a"}) + for role in (*ROLES, "lead"): + directory = runtime / "council" / run_id / role + directory.mkdir(parents=True, exist_ok=True) + directory.chmod(0o700) + snapshot["council_directories"][role] = str(directory.resolve()) + # A common-brief-only sentinel is also the actual per-role filesystem + # probe target. Peer proposals never enter another author's directory. + (directory / "brief.json").write_bytes(canonical_json(brief).encode()) + private_log = runtime / "private-logs" / f"{run_id}.boundary-probe" + private_log.parent.mkdir(parents=True, exist_ok=True) + private_log.write_bytes(b"Private coordinator boundary probe\n") + private_log.chmod(0o600) + prior_probe = store.artifact_named(run_id, "council-boundary-probe.json") + if prior_probe is None: + store.store_artifact(run_id, "council-boundary-probe.json", b'{"private_coordinator_probe":true}\n') + private_artifact = Path(store.artifact_named(run_id, "council-boundary-probe.json")["path"]) + for role in (*ROLES, "lead"): + directory = Path(snapshot["council_directories"][role]) + if fixture is None and (role != "lead" or task["lead"]["mode"] == "headless"): + routed = snapshot["routing"]["roles"][role] + snapshot["council_adapters"][role] = {} + snapshot["council_boundaries"][role] = {} + for selected in [routed["selected"], *routed["fallbacks"]]: + adapter = freeze_codex_role(selected, role=role, output_schema=output_schema(role)) + snapshot["council_adapters"][role][selected["profile_id"]] = adapter + boundary = freeze_boundary(executable=Path(adapter["binary"]), evidence=directory, native_codex=True) + forbidden = tuple(Path(path) / "brief.json" for peer, path in snapshot["council_directories"].items() if peer != role) + verify_read_boundary(boundary, own_file=directory / "brief.json", + forbidden=(*forbidden, runtime / "state.sqlite3", private_log, private_artifact)) + verify_native_bootstrap(boundary, executable=Path(adapter["binary"]), + expected_version=adapter["harness_version"], evidence=directory) + from .council_isolation import verify_native_network + verify_native_network(boundary) + snapshot["council_boundaries"][role][selected["profile_id"]] = {**boundary, "probe_status": "passed"} + snapshot["council_origin_sha256"] = digest(frozen_fields(snapshot)) + content = canonical_json(frozen_fields(snapshot)).encode() + prior = store.artifact_named(run_id, "council-origin.json") + if prior is None: + store.store_artifact(run_id, "council-origin.json", content) + elif prior["sha256"] != snapshot["council_origin_sha256"]: + raise ConflictError("recovered Council preparation changed frozen inputs") + + +def selected_for(snapshot: dict, role: str, index: int) -> dict: + route = snapshot["routing"]["roles"][role] + return [route["selected"], *route["fallbacks"]][index] + + +def validate_document(evidence: dict, snapshot: dict, attempt: dict) -> dict: + from .council import exact, text, PROMPT_VERSION, output_schema, prompt_digest + from .council_worker import role_packet + exact(evidence, {"schema_version", "role", "brief_sha256", "profile", "profile_sha256", "document", + "observed_identity", "identity_scope", "native_ids", "usage", "checks", "boundary_sha256", + "prompt_version", "prompt_sha256", "role_packet_sha256", "output_schema_sha256"}, "Council attempt") + role = attempt["role"] + selected = selected_for(snapshot, role, attempt["profile_index"]) + packet = role_packet(snapshot, role) + if (evidence["schema_version"] != 1 or evidence["role"] != role or evidence["profile"] != selected["profile"] + or evidence["profile_sha256"] != selected["profile_sha256"] + or attempt["profile_id"] != selected["profile_id"] + or evidence["brief_sha256"] != digest(snapshot["council_brief"]) + or evidence["prompt_version"] != PROMPT_VERSION + or evidence["prompt_sha256"] != prompt_digest(role, packet) + or evidence["role_packet_sha256"] != digest(packet) + or evidence["output_schema_sha256"] != digest(output_schema(role))): + raise ContractError("Council evidence differs from its actual frozen role/profile/brief") + fixture = snapshot["council_fixture"] is not None + identity = evidence["observed_identity"] + if fixture: + if (identity is not None or evidence["identity_scope"] != "all_fixture" or evidence["boundary_sha256"] is not None + or evidence["native_ids"] is not None or evidence["usage"] is not None): + raise ContractError("Council fixture cannot claim native identity/isolation") + else: + adapter = snapshot["council_adapters"][role][selected["profile_id"]] + exact(identity, {"harness", "harness_version", "model_provider", "model_id", "effort", "permission_policy", "verification"}, + "observed Council native identity") + expected_identity = {"harness": adapter["harness"], "harness_version": adapter["harness_version"], + "model_provider": adapter["model_provider"], "model_id": selected["profile"]["model_id"], + "effort": selected["profile"]["effort"]["value"], "permission_policy": "read_only", "verification": "verified"} + if (identity != expected_identity or evidence["identity_scope"] != "native_verified" + or evidence["boundary_sha256"] != snapshot["council_boundaries"][role][selected["profile_id"]]["profile_sha256"]): + raise ContractError("Council requires verified native identity and frozen read isolation") + exact(evidence["native_ids"], {"thread_id", "turn_id"}, "Council native IDs") + for key, value in evidence["native_ids"].items(): + text(value, f"Council native {key}") + if len(value) > 500: + raise ContractError("Council native IDs exceed their bound") + exact(evidence["usage"], {"input_tokens", "output_tokens", "total_tokens", "source"}, "Council native usage") + usage = evidence["usage"] + tokens = [usage[key] for key in ("input_tokens", "output_tokens", "total_tokens")] + if (usage["source"] not in {"native_reported", "unavailable"} + or any(value is not None and (type(value) is not int or value < 0) for value in tokens) + or usage["source"] == "unavailable" and any(value is not None for value in tokens) + or usage["source"] == "native_reported" and any(value is None for value in tokens)): + raise ContractError("Council native token usage is invalid or invents missing observations") + if not isinstance(evidence["checks"], list): + raise ContractError("Council trusted checks must be an array") + spec = {**snapshot["task"]["council"], "source_ids": [item["artifact_id"] for item in snapshot["council_brief"]["source_files"]]} + if role in {"proposer_a", "proposer_b"}: + validate_proposal(evidence["document"], spec) + if evidence["checks"]: + raise ContractError("Council proposers cannot supply trusted check results") + elif role == "critic": + validate_critique(evidence["document"], spec) + documents = {**snapshot["council_state"]["documents"], "critic": evidence} + verify_identities(documents, fixture=fixture) + from .workflows import validate_check_results + validate_check_results(evidence["checks"], snapshot["task"], snapshot["workspace"]) + elif role == "lead": + validate_choice(evidence["document"], spec, snapshot["council_state"]["documents"]["critic"]["document"]) + if evidence["document"]["disposition"] == "accept": + checks = snapshot["council_state"]["documents"]["critic"]["checks"] + if any(c["required_to_pass"] and c["status"] != "passed" or c.get("integrity", {}).get("status") in {"violated", "not_run"} for c in checks): + raise ContractError("Council lead cannot override failed mandatory checks/integrity") + else: + raise ContractError("Council attempt role is invalid") + return evidence + + +def add_artifact(store, run_id: str, name: str, value: Any) -> dict: + path, sha256, size = store.finalize_artifact(run_id, name, canonical_json(value).encode()) + return {"name": name, "path": path, "sha256": sha256, "byte_size": size} + + +def import_stage(store, run_id: str, attempt: dict, artifacts: list, metadata: dict, snapshot: dict, raw: bytes) -> str: + verify_origin(store, run_id, snapshot) + evidence = validate_document(decode(raw), snapshot, attempt) + role = attempt["role"] + name = f"council-{role}-{attempt['id']}.json" + artifacts.append(add_artifact(store, run_id, name, evidence)) + if role == "lead": + # The existing headless import/reopen fence remains the single lead authority. + artifacts.append(add_artifact(store, run_id, f"lead-attempt-{attempt['id']}.json", evidence)) + return store.commit_headless_lead(run_id, attempt["attempt_token"], artifacts, metadata) + updated = json.loads(canonical_json(snapshot)) + state = updated["council_state"] + state["documents"][role] = evidence + state.setdefault("artifacts", {})[role] = {"name": name, "sha256": artifacts[-1]["sha256"]} + if role != "critic": + state["next_role"] = "proposer_b" if role == "proposer_a" else "critic" + return store.commit_council_stage(run_id, attempt["attempt_token"], artifacts, metadata, updated, role) + # Save the critic projection before publishing a handoff in the same transaction. + state["next_role"] = "lead" + mapping = label_mapping(snapshot["task"]["council"]["seed"], ["proposer_a", "proposer_b"]) + checks = evidence["checks"] + failures = [c["id"] for c in checks if c["required_to_pass"] and c["status"] != "passed"] + integrity = [c["id"] for c in checks if c.get("integrity", {}).get("status") in {"violated", "not_run"}] + packet = {"schema_version": 1, "workflow": "council-decision", "candidate_sha256": snapshot["workspace"]["candidate_sha256"], + "base_oid": snapshot["base_oid"], "target_oid": snapshot["target_oid"], "brief_sha256": evidence["brief_sha256"], + "proposals": sanitized_proposals(updated), + "label_mapping": mapping, "critique": evidence["document"], "checks": checks, + "evaluation": {"accept_allowed": not failures and not integrity, "required_failures": failures, + "integrity_failures": integrity}, + "artifacts": [], "instructions": "The sole lead must retain objections and validation; votes cannot override failed checks.", + "identity_scope": "all_fixture" if snapshot["council_fixture"] is not None else "native_verified"} + bundle = {"documents": state["documents"], "label_mapping": mapping, "brief_sha256": evidence["brief_sha256"]} + bundled = add_artifact(store, run_id, f"council-evidence-{attempt['id']}.json", bundle) + artifacts.append(bundled) + packet["artifacts"] = [{"name": a["name"], "sha256": a["sha256"]} for a in (artifacts[-2], bundled)] + return store.commit_durable_handoff(run_id, attempt["attempt_token"], artifacts, metadata, packet, + mutable_snapshot=updated) + + +def reports(store, run_id: str, snapshot: dict | None, state: str, *, choice: dict | None = None, error: dict | None = None) -> dict[str, bytes]: + from .reports import _contents_with_manifest, _event_export + run = store.run(run_id) + snapshot = snapshot or {} + documents = snapshot.get("council_state", {}).get("documents", {}) + attempts = [] + for attempt in store.attempts_for_run(run_id): + role = attempt["role"] + saved = store.artifact_named(run_id, f"council-{role}-{attempt['id']}.json") + evidence = None + if saved is not None: + raw = Path(saved["path"]).read_bytes() + if hashlib.sha256(raw).hexdigest() != saved["sha256"]: + raise ConflictError("Council attempt evidence is corrupt") + evidence = decode(raw) + ended = datetime.fromisoformat(attempt["finished_at"]) if attempt["finished_at"] else datetime.now(timezone.utc) + latency = max(0, int((ended - datetime.fromisoformat(attempt["created_at"])).total_seconds() * 1000)) if attempt["pid"] is not None else None + pending = attempt["status"] in {"running", "cancelling"} and state in {"failed", "cancelled"} + metadata = json.loads(attempt["output_metadata"] or "{}") + attempts.append({"id": attempt["id"], "role": role, + "status": state if pending else "failed" if metadata.get("failure") else attempt["status"], + "profile_id": attempt["profile_id"], "profile_index": attempt["profile_index"], + "requested_profile": selected_for(snapshot, role, attempt["profile_index"])["profile"] if "routing" in snapshot else None, + "observed_identity": evidence.get("observed_identity") if evidence else None, + "usage": evidence.get("usage") if evidence else None, + "native_ids": evidence.get("native_ids") if evidence else None, "native_model_requests": None, + "worker_invocations": 1 if attempt["pid"] is not None else 0, + "policy_sha256": snapshot.get("routing", {}).get("policy", {}).get("sha256"), + "runtime_package_sha256": run.get("package_digest"), "prompt_version": "council-role-packet-v1", + "prompt_sha256": evidence.get("prompt_sha256") if evidence else None, + "role_packet_sha256": evidence.get("role_packet_sha256") if evidence else None, + "output_schema_sha256": evidence.get("output_schema_sha256") if evidence else None, + "output_sha256": saved["sha256"] if saved else None, + "latency_ms": latency, "error": error if pending else metadata.get("failure")}) + artifacts = store.artifacts_for_run(run_id) + receipt = {"schema_version": 1, "run_id": run_id, "workflow": "council-decision", "state": state, + "candidate": {"sha256": snapshot.get("workspace", {}).get("candidate_sha256"), + "base_oid": snapshot.get("base_oid"), "target_oid": snapshot.get("target_oid")}, + "completed_at": datetime.now(timezone.utc).isoformat(), "goal": snapshot.get("task", {}).get("goal"), + "candidate_sha256": snapshot.get("workspace", {}).get("candidate_sha256"), + "brief_sha256": digest(snapshot["council_brief"]) if "council_brief" in snapshot else None, + "attempts": attempts, "criteria": snapshot.get("task", {}).get("acceptance", []), + "lead": {"disposition": choice.get("disposition") if choice else None, "choice": choice, + "work_source": "headless" if snapshot.get("task", {}).get("lead", {}).get("mode") == "headless" else "host", + "usage": None}, "dissent": documents.get("critic", {}).get("document", {}).get("objections", []), + "documents": documents, "error": error, "automatic_enabled": False, + "dispositions": [choice] if choice else [], + "identity_scope": ("all_fixture" if snapshot.get("council_fixture") is not None else + "native_verified" if len(documents) == 3 else "incomplete"), + "worker_invocations": store.worker_invocations(run_id), + "execution_elapsed_ms": store._execution_elapsed_ms(run_id, run, datetime.now(timezone.utc)), + "limitations": ["Partial anonymity", "Fixture evidence establishes mechanics only"] if snapshot.get("council_fixture") is not None else ["Partial anonymity"]} + events, _, _ = _event_export(store.events_for_run(run_id)) + projected = [{key: a[key] for key in ("id", "name", "sha256", "byte_size")} for a in artifacts] + markdown = f"# DevSquad Council\n\nRun: {run_id}\nState: {state}\nAutomatic triggering: off\n\n" + canonical_json(choice or error or {}) + "\n\nDissent:\n" + canonical_json(receipt["dissent"]) + "\n" + return _contents_with_manifest(run_id, receipt, markdown, events, projected) + + +def prepared_reports(store, run_id: str, snapshot: dict | None, state: str, **kwargs) -> list: + result = [] + for name, content in reports(store, run_id, snapshot, state, **kwargs).items(): + path, sha256, size = store.finalize_artifact(run_id, name, content) + result.append({"name": name, "path": path, "sha256": sha256, "byte_size": size}) + return result + + +def decision_gate(store, run_id: str, handoff, snapshot: dict, decision: dict) -> dict: + from .workflows import validate_handoff_decision_evidence + validate_saved_handoff(store, run_id, handoff, snapshot) + validate_handoff_decision_evidence(decision, handoff.packet) + choice = decision.get("council_choice") + validate_choice(choice, snapshot["task"]["council"], handoff.packet["critique"]) + if choice["disposition"] != decision["disposition"] or choice["reason"] != decision["reason"]: + raise ContractError("Council choice contradicts the lead disposition") + if decision["disposition"] == "accept" and handoff.packet["evaluation"]["accept_allowed"] is not True: + raise ContractError("Council acceptance is blocked by mandatory checks/integrity") + if decision["disposition"] == "revise": + raise ContractError("Council additional rounds require a new explicitly capped run") + return choice + + +def finish_recorded(store, run_id: str, handoff, snapshot: dict, decision: dict) -> dict: + choice = decision_gate(store, run_id, handoff, snapshot, decision) + entry = store.recorded_handoff_submission(run_id, handoff.handoff_id) + if entry is None or entry["decision"] != decision: + raise ConflictError("Council recorded decision differs from continuation") + if store.run(run_id)["state"] in {"succeeded", "failed"}: + return {"run_id": run_id, "state": store.run(run_id)["state"], "replayed": True, + "disposition": decision["disposition"], "launched": False} + terminal = "succeeded" if decision["disposition"] == "accept" else "failed" + artifacts = prepared_reports(store, run_id, snapshot, terminal, choice=choice) + result = store.complete_handoff_terminal(run_id, handoff.handoff_id, decision["submission_id"], + decision["submission_hash"], artifacts, terminal, {"disposition": decision["disposition"], "receipt": "result-receipt.json"}) + return {"run_id": run_id, **result, "disposition": decision["disposition"], "launched": False} + + +def complete(service, store, run_id: str, handoff, snapshot: dict, decision: dict) -> dict: + claim = service._decode_claim(decision.pop("_claim")) + decision_gate(store, run_id, handoff, snapshot, decision) + submission = store.record_handoff_submission(run_id, claim, decision) + result = finish_recorded(store, run_id, handoff, snapshot, decision) + result["replayed"] = submission.replayed + return result + + +def saved_claim(store, run_id: str, owner: str): + """Reuse exactly our durable live claim, or reacquire only our expired one.""" + handoff = store.handoff_snapshot(run_id) + prior = handoff.claim + if prior is not None and prior.owner_id != owner: + raise ConflictError("Council continuation cannot take another host's claim") + if prior is not None and datetime.now(timezone.utc) < datetime.fromisoformat(prior.expires_at): + return prior + return store.claim_handoff(run_id, store.run(run_id)["version"], owner) + + +def continue_host_intent(service, store, run_id: str, handoff, snapshot: dict) -> dict: + decision = store.terminal_finish_decision(run_id, handoff.handoff_id) + if decision is None: + raise ConflictError("Council host continuation has no exact current guided-finish claim marker") + decision_gate(store, run_id, handoff, snapshot, decision) + # Shared R6 authority reuses only the latest canonical guided marker. An + # expired identical intent gets a fresh fence; same-name app claims do not. + claim = store.claim_handoff(run_id, store.run(run_id)["version"], "terminal-operator", + initial_only=True, terminal_decision=decision) + return complete(service, store, run_id, handoff, snapshot, + {**decision, "_claim": service._claim_payload(claim)}) + + +def continue_lead(service, store, run_id: str, run: dict, handoff, snapshot: dict) -> dict: + entry = store.recorded_handoff_submission(run_id, handoff.handoff_id) + if entry is not None: + return finish_recorded(store, run_id, handoff, snapshot, entry["decision"]) + if snapshot["task"]["lead"]["mode"] != "headless": + return continue_host_intent(service, store, run_id, handoff, snapshot) + verify_origin(store, run_id, snapshot) + leads = [a for a in store.attempts_for_run(run_id) if a["role"] == "lead"] + for attempt in reversed(leads): + artifact = store.artifact_named(run_id, f"lead-attempt-{attempt['id']}.json") + if artifact is None: + continue + raw = Path(artifact["path"]).read_bytes() + if hashlib.sha256(raw).hexdigest() != artifact["sha256"]: + raise ConflictError("Council lead artifact is corrupt") + evidence = validate_document(decode(raw), snapshot, attempt) + choice = evidence["document"] + claim = saved_claim(store, run_id, f"headless-lead:{attempt['id']}") + body = {"schema_version": 1, "submission_id": f"headless-{attempt['id']}-{claim.fencing_token}", + "disposition": choice["disposition"], "reason": choice["reason"], "council_choice": choice, + "evidence_refs": [{"artifact_id": ref["artifact_id"], "sha256": ref["sha256"]} for ref in handoff.packet["artifacts"]]} + return complete(service, store, run_id, handoff, snapshot, + {**body, "submission_hash": request_hash(body), "_claim": service._claim_payload(claim)}) + queued = store.queue_headless_lead(run_id, run["version"]) + if queued["action"] == "budget_exhausted": + error = {"error": "BUDGET_EXHAUSTED", "message": "Council lead budget exhausted"} + prepared = prepared_reports(store, run_id, snapshot, "failed", error=error) + store.fail_queued_budget(run_id, run["version"], error, prepared) + return {"run_id": run_id, "state": "failed", "launched": False} + current = store.run(run_id) + package, package_digest = service._verified_package(current) + return {"run_id": run_id, "state": current["state"], "launched": True, + "launch": (current["version"], package, package_digest)} diff --git a/plugin/core/src/devsquad/council_task_entry.py b/plugin/core/src/devsquad/council_task_entry.py new file mode 100644 index 0000000..7538154 --- /dev/null +++ b/plugin/core/src/devsquad/council_task_entry.py @@ -0,0 +1,74 @@ +"""Manual, bounded Council command preparation; automatic triggering is off.""" +from __future__ import annotations + +from pathlib import Path + +from .contracts import CapabilityUnavailable, ContractError +from .council import ROLES, digest, validate_spec +from .task_entry import _relative_path, _resolve_commit, resolve_repository +from .validation import validate_task + + +def build_council_task(*, project_dir: Path, goal: str, identities: list[dict], + base_ref: str = "HEAD", target_ref: str = "HEAD", read_paths=(), checks=(), + lead_mode: str = "headless", max_invocations: int = 4, evidence=(), rubric=None) -> tuple[dict, dict]: + if not isinstance(goal, str) or not goal.strip() or len(goal) > 16000: + raise ContractError("Council question must be non-empty and bounded") + if (len(identities) != 3 or any(not isinstance(value, dict) or value.get("harness") != "codex" for value in identities) + or len({value.get("model_id", "").casefold() for value in identities}) != 3): + raise CapabilityUnavailable("Council needs three distinct entitled, verified Codex model IDs") + repo = resolve_repository(project_dir) + base, target = _resolve_commit(repo, base_ref), _resolve_commit(repo, target_ref) + paths = list(dict.fromkeys(_relative_path(path, "Council read path") for path in read_paths)) + profiles = [] + for role, identity in zip(ROLES, identities): + if any(not isinstance(identity.get(key), str) or not identity[key] for key in ("model_id", "model_family", "effort", "harness_version")): + raise ContractError("Council native catalog identity is incomplete") + profile = {"id": f"managed-council-{role}", "harness": "codex", "model_family": identity["model_family"], + "model_id": identity["model_id"], "effort": {"value": identity["effort"], "transport": "native"}, + "required_tools": [], "permission_policy": "read_only", "account_pool_id": identity.get("account_pool_id", "codex-subscription"), + "billing_mode": "subscription", "quality_status": "trial", + "evidence_refs": [f"runtime-catalog:{identity['harness_version']}:{identity['model_id']}", + *([f"runtime-catalog-fingerprint:{identity['catalog_fingerprint']}"] if "catalog_fingerprint" in identity else []), + *([f"runtime-native-scope:{identity['native_scope']}"] if "native_scope" in identity else [])]} + profile["id"] += "-" + digest(profile)[:16] + profiles.append(profile) + rubric = rubric or [{"id": "correctness", "description": "Claims fit the frozen evidence"}, + {"id": "risk", "description": "Important risks and objections are retained"}, + {"id": "validation", "description": "Decision names objective validation and uncertainty"}] + seed = digest({"goal": goal.strip(), "base_oid": base, "target_oid": target, "read_paths": paths, + "evidence": list(evidence), "rubric": rubric, "profiles": profiles}) + council = {"schema_version": 1, "enabled": True, "automatic": False, + "reason": "Explicit manual Council request", "min_valid_proposals": 2, "required_critics": 1, + "max_invocations": max_invocations, "seed": seed, "evidence": list(evidence), "rubric": rubric} + validate_spec(council) + roles = {role: [{"kind": "profile", "id": profile["id"]}] for role, profile in zip(ROLES, profiles)} + if lead_mode == "headless": + roles["lead"] = [{"kind": "profile", "id": profiles[-1]["id"]}] + declared_checks = [{"id": "candidate-integrity", "argv": ["git", "diff", "--check", base, target], + "cwd": ".", "timeout_seconds": 30, "required_to_pass": True}] + declared_checks.extend({"id": f"user-check-{index + 1}", "argv": list(argv), "cwd": ".", + "timeout_seconds": 60, "required_to_pass": True} for index, argv in enumerate(checks)) + task = {"schema_version": 1, "project": {"repo_path": str(repo), "base_ref": base, "target_ref": target}, + "workflow": "council-decision", "goal": goal.strip(), "task_class": "managed-council", + "acceptance": [{"id": item["id"], "description": item["description"], "evidence_kind": "review"} for item in rubric], + "checks": declared_checks, "scope": {"read_paths": paths, "write_paths": []}, "lead": {"mode": lead_mode}, + "routing": {"profiles": {"schema_version": 1, "profiles": profiles, "bindings": {}}, + "policy": {"schema_version": 1, "id": "managed-manual-council", "version": 1, + "roles": roles, "task_classes": {"managed-council": "trial"}, "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, "account_pools": {profile["account_pool_id"]: { + "allowed_billing_modes": ["subscription"], "max_concurrency": 1, "unknown_capacity_policy": "allow_bounded"} for profile in profiles}, + "experiment_budget": {}, "decision_helper": {"schema_version": 1, "mode": "off"}}}, + "budget": {"wall_seconds": 600, "max_worker_invocations": max_invocations, "max_revisions": 0, "max_fallbacks_per_step": 0}, + "origin": {"surface": "cli"}, "council": council} + validate_task(task, require_existing_repo=True) + summary = {"workflow": "council-decision", "project": str(repo), "base_oid": base, "target_oid": target, "goal": task["goal"], + "profiles": {role: {"profile_id": profile["id"], "model_id": profile["model_id"], "quality_status": "trial"} for role, profile in zip(ROLES, profiles)}, + "planned_roles": {role: {"profile_id": profile["id"], "harness": "codex", "model_id": profile["model_id"], + "effort": profile["effort"]["value"], "selection_mode": "catalog_trial", "quality_status": "trial"} for role, profile in zip(ROLES, profiles)}, + "selection_reason": "Three distinct available catalog identities; actual entitlement and identity must be verified before quorum; no quality qualification implied", + "scope": task["scope"], "checks": [item["id"] for item in declared_checks], "check_plan": declared_checks, "task_sha256": digest(task), + "budget": task["budget"], "lead": task["lead"], "automatic_enabled": False, + "native_ready": False, + "limitations": ["Manual only", "One round", "Trial catalog profiles are not quality proof", "Requires verified macOS default-deny isolation"]} + return task, summary diff --git a/plugin/core/src/devsquad/council_worker.py b/plugin/core/src/devsquad/council_worker.py new file mode 100644 index 0000000..99acada --- /dev/null +++ b/plugin/core/src/devsquad/council_worker.py @@ -0,0 +1,97 @@ +"""One Council role, executed inside the existing gated durable attempt.""" +from __future__ import annotations + +import json +from pathlib import Path +import sys +import time + +from .contracts import ContractError +from .council import MAX_PACKET_BYTES, PROMPT_VERSION, digest, sanitized_proposals, output_schema, role_prompt, prompt_digest +from .council_runtime import selected_for +from .store import canonical_json + + +def role_packet(snapshot: dict, role: str) -> dict: + packet = {"brief": snapshot["council_brief"]} + state = snapshot["council_state"] + if role in {"critic", "lead"}: + if not {"proposer_a", "proposer_b"}.issubset(state["documents"]): + raise ContractError("Council proposal barrier has not completed") + packet["proposals"] = sanitized_proposals(snapshot) + if role == "lead": + packet["critique"] = state["documents"]["critic"]["document"] + packet["checks"] = state["documents"]["critic"]["checks"] + return packet + + +def run(snapshot: dict) -> dict: + role = snapshot["council_role"] + selected = selected_for(snapshot, role, snapshot["council_profile_index"]) + packet = role_packet(snapshot, role) + directory = Path(snapshot["council_directories"][role]) + # The role directory contains only its frozen/sanitized packet. Coordinator + # logs, ledger, raw provenance and all peer directories are outside the jail. + packet_path = directory / "evidence.json" + encoded = canonical_json(packet).encode() + if len(encoded) > MAX_PACKET_BYTES: + raise ContractError("Council role packet exceeds its byte cap") + if packet_path.exists() and packet_path.read_bytes() != encoded: + raise ContractError("Council role packet changed after finalization") + packet_path.write_bytes(encoded) + fixture = snapshot["council_fixture"] + boundary_sha256 = None + if fixture is not None: + value = fixture.get(role) + if not isinstance(value, dict): + raise ContractError(f"Council fixture is missing {role}") + if value.get("delay_seconds"): + time.sleep(value["delay_seconds"]) + if value.get("error"): + raise ContractError(value["error"]) + result = {"document": value.get("document"), "observed_identity": None, "native_ids": None, "usage": None} + identity_scope = "all_fixture" + else: + from .codex_review_worker import run as native_turn + adapter = snapshot["council_adapters"][role][selected["profile_id"]] + boundary = snapshot["council_boundaries"][role][selected["profile_id"]] + native = {"task": snapshot["task"], "workspace": {"path": str(directory)}, + "routing": {"roles": {role: {"selected": selected}}}, + "council_adapter": adapter, "council_boundary": boundary} + prompt = role_prompt(role, packet) + result = native_turn(native, council_role=role, council_prompt=prompt) + identity_scope = "native_verified" + boundary_sha256 = boundary["profile_sha256"] + checks = [] + if role == "critic": + from .review_worker import run_review_and_checks + from .workflows import review_mode + # Reuse the trusted, isolated candidate-check path. This temporary + # internal review document is only a check driver, never Council review evidence. + check_snapshot = json.loads(canonical_json(snapshot)) + check_snapshot["task"]["workflow"] = "branch-review" + check_snapshot["task"].pop("council", None) + check_snapshot["routing"]["roles"]["reviewer"] = {"selected": selected} + workspace = snapshot["workspace"] + driver = {"schema_version": 1, "candidate_sha256": workspace["candidate_sha256"], + "base_oid": workspace["base_oid"], "target_oid": workspace["target_oid"], + "review_mode": review_mode(check_snapshot["task"]), "verdict": "clean", "summary": "Trusted checks only", "findings": []} + checks = run_review_and_checks(check_snapshot, driver)["checks"] + return {"schema_version": 1, "role": role, "brief_sha256": digest(snapshot["council_brief"]), + "prompt_version": PROMPT_VERSION, "prompt_sha256": prompt_digest(role, packet), + "role_packet_sha256": digest(packet), "output_schema_sha256": digest(output_schema(role)), + "profile": selected["profile"], "profile_sha256": selected["profile_sha256"], + **result, "identity_scope": identity_scope, "checks": checks, "boundary_sha256": boundary_sha256} + + +def main() -> int: + payload = sys.stdin.buffer.read(4 * MAX_PACKET_BYTES + 1) + if len(payload) > 4 * MAX_PACKET_BYTES: + raise ContractError("Council worker snapshot exceeds its byte cap") + snapshot = json.loads(payload.decode("utf-8")) + sys.stdout.write(canonical_json(run(snapshot)) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugin/core/src/devsquad/detached.py b/plugin/core/src/devsquad/detached.py index 48dc5f7..4bc24a7 100644 --- a/plugin/core/src/devsquad/detached.py +++ b/plugin/core/src/devsquad/detached.py @@ -109,18 +109,22 @@ def main(argv=None): else "implementer" if delivery_implementer else "reviewer" ) + if workflow == "council-decision": + from .council_runtime import verify_origin + verify_origin(store, args.run_id, snapshot) + role = "lead" if headless_lead else snapshot["council_state"]["next_role"] workflow_role = ( "internal_fake_delay" not in snapshot - and workflow in {"branch-review", "issue-delivery"} + and workflow in {"branch-review", "issue-delivery", "council-decision"} ) profile_index = None profile_id = None if workflow_role: routed_role = snapshot["routing"]["roles"][role] candidates = [routed_role["selected"], *routed_role["fallbacks"]] - profile_index = _profile_index( - store, args.run_id, role, handoff, - ) + profile_index = (sum(a.get("role") == role and _is_saved_fallback_failure(a) + for a in store.attempts_for_run(args.run_id)) + if workflow == "council-decision" else _profile_index(store, args.run_id, role, handoff)) if profile_index >= len(candidates): raise ConflictError("frozen role fallback set is exhausted") attempt_selection = candidates[profile_index] @@ -130,13 +134,18 @@ def main(argv=None): "implementer": "implementation_adapter", "reviewer": "review_adapter", "lead": "lead_adapter", + "proposer_a": "council_adapter", "proposer_b": "council_adapter", "critic": "council_adapter", }[role] adapters_key = { "implementer": "implementation_adapters", "reviewer": "review_adapters", "lead": "lead_adapters", + "proposer_a": "council_adapters", "proposer_b": "council_adapters", "critic": "council_adapters", }[role] adapters = snapshot.get(adapters_key) + if workflow == "council-decision": + adapter_key = "council_adapter" + adapters = snapshot["council_adapters"].get(role, {}) adapter = ( adapters.get(profile_id) if isinstance(adapters, dict) @@ -154,7 +163,9 @@ def main(argv=None): selected["account_pool_id"], "verified" if adapter else "unknown", ) - if role == "implementer": + if workflow == "council-decision": + module = "devsquad.council_worker" + elif role == "implementer": module = ( "devsquad.claude_delivery_worker" if adapter else "devsquad.delivery_worker" @@ -171,7 +182,11 @@ def main(argv=None): ) command = [sys.executable, "-P", "-m", module] worker_snapshot = json.loads(canonical_json(snapshot)) - worker_snapshot["routing"]["roles"][role]["selected"] = attempt_selection + if workflow == "council-decision": + worker_snapshot["council_role"] = role + worker_snapshot["council_profile_index"] = profile_index + else: + worker_snapshot["routing"]["roles"][role]["selected"] = attempt_selection if adapter is None: worker_snapshot.pop(adapter_key, None) else: @@ -185,6 +200,7 @@ def main(argv=None): input_path, _, _ = store.finalize_artifact( args.run_id, ( + f"council-input-{role}-{profile_index}.json" if workflow == "council-decision" else f"lead-workflow-input-{handoff.sequence}-{profile_index}.json" if headless_lead else f"implementation-input-{profile_index}.json" @@ -205,6 +221,8 @@ def main(argv=None): args.run_id, args.expected_version, ) return 1 + if workflow == "council-decision": + environment["DEVSQUAD_COUNCIL_ROLE"] = role spec = LaunchSpec( 1, identity.harness, diff --git a/plugin/core/src/devsquad/diagnostics.py b/plugin/core/src/devsquad/diagnostics.py index 580ddc5..b447afb 100644 --- a/plugin/core/src/devsquad/diagnostics.py +++ b/plugin/core/src/devsquad/diagnostics.py @@ -5,6 +5,7 @@ import json import os from pathlib import Path +import platform import re import selectors import shutil @@ -28,6 +29,7 @@ capture_probe_identity, close_probe as _close_probe, subscription_environment as _environment, + wait_probe_exit, ) SOURCE_ROOT = Path(__file__).resolve().parents[2] @@ -75,6 +77,7 @@ def _probe_output( if deadline - time.monotonic() <= 0: raise TimeoutError("diagnostic probe timed out") output = chunks.decode("utf-8") + wait_probe_exit(process, start_identity=start_identity, deadline=deadline) finally: # Retain the direct child's PID until group cleanup is confirmed; # waiting here first would discard the exited-parent ownership anchor. @@ -299,7 +302,9 @@ def build_doctor_report( "required_adapters": list(required), "blocked_adapters": blocked, } supported_workflows["council"] = { - "supported": False, "ready": False, "reason": "not_implemented", + "supported": True, "implemented_partial": True, "ready": False, + "native_ready": False, "automatic_enabled": False, + "reason": "native_network_attestation_unavailable", } workflow_ready = any(row["ready"] for row in supported_workflows.values()) local_apps_required = bool(installed_apps) @@ -329,4 +334,13 @@ def build_doctor_report( }, "adapters": adapters, "local_apps": local_apps, + "workflows": {"council-decision": { + "implemented": True, "automatic_enabled": False, "ready": False, + "readiness": "native_network_attestation_unavailable", + "native_network_ready": False, + "read_boundary_available": platform.system() == "Darwin" and Path("/usr/bin/sandbox-exec").is_file(), + "requirements": ["three distinct entitled verified read-only model IDs", "frozen default-deny Seatbelt probe", + "subscription/network compatibility", "two valid independent proposals and one distinct critic"], + "supported_rounds": 1, + }}, } diff --git a/plugin/core/src/devsquad/learning.py b/plugin/core/src/devsquad/learning.py index d50e888..07d494e 100644 --- a/plugin/core/src/devsquad/learning.py +++ b/plugin/core/src/devsquad/learning.py @@ -22,7 +22,7 @@ SELECTION_MODES = {"automatic", "pinned", "experimental"} CRITERION_STATES = {"passed", "failed", "unknown"} CONTRIBUTION_RESULTS = {"failed", "successful", "repair", "finding", "neutral"} -ROLES = {"worker", "implementer", "reviewer", "lead", "researcher"} +ROLES = {"worker", "implementer", "reviewer", "lead", "researcher", "proposer_a", "proposer_b", "critic"} MAX_CLOCK_SKEW = timedelta(minutes=5) EXPERIMENT_FIELDS = { "schema_version", "experiment_id", "project_path", "question", "hypothesis", diff --git a/plugin/core/src/devsquad/mcp_server.py b/plugin/core/src/devsquad/mcp_server.py index 6c00cea..4b138c6 100644 --- a/plugin/core/src/devsquad/mcp_server.py +++ b/plugin/core/src/devsquad/mcp_server.py @@ -105,6 +105,13 @@ def _task_with_bound_origin(self, task: dict[str, Any]) -> dict[str, Any]: bound["origin"] = origin return bound + def _saved_read_response(self, operation: Callable[[], dict[str, Any]]) -> dict[str, Any]: + def guarded(): + if self.environment.get("DEVSQUAD_COUNCIL_ROLE", ""): + raise PolicyDenied("Council worker sessions cannot read saved team artifacts, events or handoff provenance") + return operation() + return self._response(guarded) + def start( self, task: dict[str, Any], @@ -129,14 +136,14 @@ def doctor(self) -> dict[str, Any]: def status(self, run_id: str) -> dict[str, Any]: """Inspect the current projection for a saved run.""" - return self._response(lambda: self.service.status(run_id)) + return self._saved_read_response(lambda: self.service.status(run_id)) def events( self, run_id: str, after: int = 0, limit: int = 100, ) -> dict[str, Any]: """Read one bounded event page using its durable integer cursor.""" - return self._response(lambda: self.service.events(run_id, after, limit)) + return self._saved_read_response(lambda: self.service.events(run_id, after, limit)) def result( self, run_id: str, preview_bytes: int = 4096, @@ -180,7 +187,7 @@ def operation() -> dict[str, Any]: artifacts.append(artifact) return {**result, "artifacts": artifacts, "preview_bytes": preview_bytes} - return self._response(operation) + return self._saved_read_response(operation) def cancel(self, run_id: str) -> dict[str, Any]: """Persist cancellation intent without observing worker completion.""" diff --git a/plugin/core/src/devsquad/migrations/017_council_contract_epoch.sql b/plugin/core/src/devsquad/migrations/017_council_contract_epoch.sql new file mode 100644 index 0000000..5f412c4 --- /dev/null +++ b/plugin/core/src/devsquad/migrations/017_council_contract_epoch.sql @@ -0,0 +1,5 @@ +-- Contract epoch only: Council's read/claim/decision authority is not safe for +-- schema-16 packages sharing this ledger. Existing exclusive migration, +-- active/recoverable deferral and per-connection write triggers fence them. +-- No table, scheduling service or historical outcome representation changes. +SELECT 1; diff --git a/plugin/core/src/devsquad/probe_process.py b/plugin/core/src/devsquad/probe_process.py index 77ef18a..f97ee3a 100644 --- a/plugin/core/src/devsquad/probe_process.py +++ b/plugin/core/src/devsquad/probe_process.py @@ -166,6 +166,41 @@ def child_live(value: int | None = None) -> bool: time.sleep(min(0.02, max(0, deadline - time.monotonic()))) +def wait_probe_exit( + process: subprocess.Popen[Any], *, start_identity: str | None, deadline: float, +) -> None: + """Observe natural completion without reaping the group ownership anchor. + + EOF does not imply child exit. Share the caller's probe deadline and use + WNOWAIT, or the exact retained zombie-child observation on older macOS, + before cleanup is allowed to terminate/reap the process group. + """ + if start_identity is None or not isinstance(process, _POPEN_TYPE): + raise ContractError("diagnostic process ownership is unavailable") + while True: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("diagnostic probe timed out") + with _reap_guard(process, deadline=deadline): + if process.returncode is not None: + raise ContractError("diagnostic process ownership anchor was already reaped") + observed = process_start_identity(process.pid) + if observed is not None and observed != start_identity: + raise ContractError("diagnostic process identity changed before natural exit") + waitid = getattr(os, "waitid", None) + if callable(waitid): + try: + result = waitid(os.P_PID, process.pid, os.WEXITED | os.WNOHANG | os.WNOWAIT) + except (OSError, AttributeError) as exc: + raise ContractError("diagnostic process ownership is unavailable") from exc + exited = result is not None and result.si_pid == process.pid + else: + exited = _retained_child_anchor(process, deadline=deadline, locked=True) + if exited: + return + time.sleep(min(0.02, max(0, deadline - time.monotonic()))) + + def close_probe(process: subprocess.Popen[Any], *, start_identity: str | None) -> None: """Stop the owned group, confirm no live members, reap and close streams. diff --git a/plugin/core/src/devsquad/reports.py b/plugin/core/src/devsquad/reports.py index f406869..b0270d7 100644 --- a/plugin/core/src/devsquad/reports.py +++ b/plugin/core/src/devsquad/reports.py @@ -139,7 +139,7 @@ def build_handoff_reports( raise ContractError("handoff report packet hash is invalid") json_name, markdown_name = handoff_report_names(sequence) workflow = packet.get("workflow", "branch-review") - if workflow not in {"branch-review", "issue-delivery"}: + if workflow not in {"branch-review", "issue-delivery", "council-decision"}: raise ContractError("handoff report workflow is invalid") report = { "schema_version": 1, @@ -158,6 +158,9 @@ def build_handoff_reports( "packet": packet, "next_action": "claim_handoff", } + if workflow == "council-decision": + return {json_name: (canonical_json(report) + "\n").encode(), + markdown_name: (f"# DevSquad Council handoff\n\nRun: {run_id}\n\n" + canonical_json(packet) + "\n").encode()} review = packet.get("review") if isinstance(packet.get("review"), dict) else {} lines = [ f"# DevSquad {workflow} handoff", diff --git a/plugin/core/src/devsquad/router.py b/plugin/core/src/devsquad/router.py index 1ac5af0..fbb2fd3 100644 --- a/plugin/core/src/devsquad/router.py +++ b/plugin/core/src/devsquad/router.py @@ -23,10 +23,12 @@ "reviewer": "read_only", "lead": "read_only", "researcher": "read_only", + "proposer_a": "read_only", "proposer_b": "read_only", "critic": "read_only", } WORKFLOW_ROLES = { "branch-review": ("reviewer",), "issue-delivery": ("implementer", "reviewer"), + "council-decision": ("proposer_a", "proposer_b", "critic"), } CAPACITY_STATES = {"available", "exhausted", "unknown"} CAPACITY_EVIDENCE_FIELDS = { @@ -361,6 +363,7 @@ def resolve_routing( selected_roles: dict[str, Any] = {} implementer = None + council_models = set() max_fallbacks = task["budget"]["max_fallbacks_per_step"] for role in roles: eligible, excluded, has_static_candidate = _policy_candidates( @@ -372,6 +375,15 @@ def resolve_routing( capacity, implementer, ) + if role in {"proposer_a", "proposer_b", "critic"}: + reused = [item for item in eligible if item["profile"]["model_id"].casefold() in council_models] + excluded.extend({"profile_id": item["profile_id"], "reason": "Council model identity already allocated"} for item in reused) + eligible = [item for item in eligible if item not in reused] + # Distinct model IDs are mandatory; qualified cross-family choices + # are a preference, not a requirement for a third provider. + families = {entry["selected"]["profile"]["model_family"].casefold() for key, entry in selected_roles.items() + if key in {"proposer_a", "proposer_b", "critic"}} + eligible.sort(key=lambda item: item["profile"]["model_family"].casefold() in families) override = overrides.get(role) selected = None source = "automatic" @@ -384,6 +396,8 @@ def resolve_routing( raise ProfileUnsupported( f"pinned {role} profile does not exist: {override['profile_id']}" ) + if role in {"proposer_a", "proposer_b", "critic"} and pinned["model_id"].casefold() in council_models: + raise ProfileUnsupported("Council override repeats a proposer/critic model identity") static_reason = _static_reason( pinned, role, @@ -434,6 +448,9 @@ def resolve_routing( } if role == "implementer": implementer = selected["profile"] + if role in {"proposer_a", "proposer_b", "critic"}: + council_models.update(item["profile"]["model_id"].casefold() for item in + [selected_roles[role]["selected"], *selected_roles[role]["fallbacks"]]) return { "schema_version": 1, diff --git a/plugin/core/src/devsquad/service.py b/plugin/core/src/devsquad/service.py index 57a4905..05c43f7 100644 --- a/plugin/core/src/devsquad/service.py +++ b/plugin/core/src/devsquad/service.py @@ -119,6 +119,9 @@ def _preparation_failure_artifacts( snapshot: dict[str, Any] | None, error: dict[str, Any], ) -> list[dict[str, Any]] | None: + if task.get("workflow") == "council-decision": + from .council_runtime import prepared_reports + return prepared_reports(store, run_id, snapshot, "failed", error=error) if task.get("workflow") not in {"branch-review", "issue-delivery"}: return None reports = build_early_terminal_reports( @@ -250,6 +253,9 @@ def _paused_review_terminal_artifacts( error: dict[str, Any] | None, ) -> list[dict[str, Any]]: """Materialize complete M3 reports before a paused run terminalizes.""" + if snapshot.get("task", {}).get("workflow") == "council-decision": + from .council_runtime import prepared_reports + return prepared_reports(store, run_id, snapshot, state, error=error) attempts, dispositions = self._saved_review_progress( store, run_id, snapshot, handoff, ) @@ -290,6 +296,12 @@ def fail_budget_exhausted( try: run = store.run(run_id) snapshot = self._review_snapshot(run) + if snapshot.get("task", {}).get("workflow") == "council-decision": + from .council_runtime import prepared_reports + error = {"error": "BUDGET_EXHAUSTED", "message": "Council run budget is exhausted"} + artifacts = prepared_reports(store, run_id, snapshot, "failed", error=error) + version = store.fail_queued_budget(run_id, expected_version, error, artifacts) + return {"run_id": run_id, "state": "failed", "version": version} if snapshot.get("task", {}).get("workflow") not in { "branch-review", "issue-delivery", }: @@ -595,6 +607,7 @@ def _resolve_snapshot( internal_lead_fixture: dict[str, Any] | None = None, internal_implementation_fixture: dict[str, Any] | None = None, internal_decision_fixture: dict[str, Any] | None = None, + internal_council_fixture: dict[str, Any] | None = None, preparation_fencing_token: int | None = None, capacity_store: Store | None = None, ) -> dict[str, Any]: @@ -690,7 +703,7 @@ def _resolve_snapshot( ) if decision_observation is not None: snapshot["decision_observation"] = decision_observation - if task["workflow"] == "branch-review": + if task["workflow"] in {"branch-review", "council-decision"}: snapshot["workspace"] = prepare_review_workspace( repo, self.runtime, @@ -710,6 +723,10 @@ def _resolve_snapshot( scope_paths, required_clean_paths=config_paths.values(), ) + if task["workflow"] == "council-decision": + from .council_runtime import prepare + prepare(snapshot, store=capacity_store, run_id=run_id, runtime=self.runtime, + fixture=internal_council_fixture) else: snapshot["delivery_workspace"] = prepare_delivery_workspace( repo, @@ -816,11 +833,12 @@ def _continue_preparation( internal_decision_fixture = submitted.get( "_internal_decision_fixture" ) + internal_council_fixture = submitted.get("_internal_council_fixture") store.validate_predecessor(run_id, fencing_token, supersedes_run_id) validated_supersedes_run_id = supersedes_run_id validate_task(task, require_existing_repo=True) minimum_headless_invocations = ( - 3 if task["workflow"] == "issue-delivery" else 2 + 4 if task["workflow"] == "council-decision" else 3 if task["workflow"] == "issue-delivery" else 2 ) if (task["lead"]["mode"] == "headless" and task["budget"]["max_worker_invocations"] @@ -842,6 +860,7 @@ def _continue_preparation( internal_lead_fixture=internal_lead_fixture, internal_implementation_fixture=internal_implementation_fixture, internal_decision_fixture=internal_decision_fixture, + internal_council_fixture=internal_council_fixture, preparation_fencing_token=fencing_token, capacity_store=store, ) @@ -878,7 +897,7 @@ def _continue_preparation( snapshot["review_adapter"] = snapshot["review_adapters"][ reviewer_route["selected"]["profile_id"] ] - if (internal_delay is None and task["lead"]["mode"] == "headless" + if (internal_delay is None and task["lead"]["mode"] == "headless" and task["workflow"] != "council-decision" and internal_lead_fixture is None): lead_route = snapshot["routing"]["roles"]["lead"] lead_candidates = [lead_route["selected"], *lead_route["fallbacks"]] @@ -960,8 +979,14 @@ def start( _internal_lead_fixture: dict[str, Any] | None = None, _internal_implementation_fixture: dict[str, Any] | None = None, _internal_decision_fixture: dict[str, Any] | None = None, + _internal_council_fixture: dict[str, Any] | None = None, ) -> dict[str, Any]: validate_task(task, require_existing_repo=True) + if task["workflow"] == "council-decision" and any(value is not None for value in + (_internal_fake_delay, _internal_review_fixture, _internal_lead_fixture, _internal_implementation_fixture)): + raise ContractError("Council requires its explicit all-fixture seam or native roles") + if _internal_council_fixture is not None and task["workflow"] != "council-decision": + raise ContractError("Council fixture requires council-decision") if trial is not None: from .learning import validate_experiment from .experiment_provenance import assignment_for @@ -1009,6 +1034,8 @@ def start( ) if _internal_decision_fixture is not None: submitted["_internal_decision_fixture"] = _internal_decision_fixture + if _internal_council_fixture is not None: + submitted["_internal_council_fixture"] = _internal_council_fixture store = self._store() try: claim = store.claim_start(Path(task["project"]["repo_path"]), idempotency_key, submitted, f"preflight:{os.getpid()}", objective_outcome=True) @@ -1289,7 +1316,9 @@ def status(self, run_id: str) -> dict[str, Any]: try: snapshot = self._review_snapshot(run) next_action = ( - "resume_candidate_review" + "resume_council_stage" + if snapshot.get("task", {}).get("workflow") == "council-decision" + else "resume_candidate_review" if snapshot.get("task", {}).get("workflow") == "issue-delivery" and isinstance(snapshot.get("candidate"), dict) and (handoff is None or handoff.status != "open") @@ -1344,6 +1373,8 @@ def handoff_view(self, run_id: str) -> dict[str, Any]: run, _, handoff = store.status_snapshot(run_id) if handoff is None or handoff.status != "open": raise ConflictError("run has no current open handoff to inspect") + if self._review_snapshot(run).get("task", {}).get("workflow") == "council-decision": + return self.council_handoff_view(run_id) packet = validate_saved_review_handoff(handoff.packet, self._review_snapshot(run)) report = store.artifact_named(run_id, handoff_report_names(handoff.sequence)[1]) if report is None: @@ -1365,7 +1396,10 @@ def handoff_view(self, run_id: str) -> dict[str, Any]: finally: store.close() - def finish(self, run_id: str, disposition: str, reason: str) -> dict[str, Any]: + def finish(self, run_id: str, disposition: str, reason: str, *, chosen: str | None = None, + supported_claims: list[str] | None = None, + discarded_alternatives: list[str] | None = None, + validation: str | None = None) -> dict[str, Any]: """Guided terminal host disposition over the existing fenced handoff gates.""" if disposition not in {"accept", "reject", "revise"}: raise ContractError("finish disposition is invalid") @@ -1378,6 +1412,14 @@ def finish(self, run_id: str, disposition: str, reason: str) -> dict[str, Any]: or handoff is None or handoff.status != "open"): raise ConflictError("run has no current open handoff to finish; inspect squad status RUN") snapshot = self._review_snapshot(run) + if snapshot.get("task", {}).get("workflow") == "council-decision": + if chosen is None or validation is None: + raise ContractError("Council finish requires explicit --choose and --validation; inspect squad status RUN") + return self.finish_council(run_id, disposition, reason, chosen=chosen, + supported_claims=supported_claims or [], discarded_alternatives=discarded_alternatives or [], + validation=validation) + if any(value is not None for value in (chosen, supported_claims, discarded_alternatives, validation)): + raise ContractError("Council choice flags cannot be used for a review or delivery handoff") if snapshot.get("task", {}).get("lead", {}).get("mode") != "host": raise ConflictError("headless lead owns this handoff; run squad resume RUN") packet = validate_saved_review_handoff(handoff.packet, snapshot) @@ -1441,6 +1483,16 @@ def cancel(self, run_id: str) -> dict[str, Any]: store = self._store() try: run = store.run(run_id) + if run["state"] == "queued" and run["phase"] is None: + try: + snapshot = self._review_snapshot(run) + except ConflictError: + snapshot = {} + if snapshot.get("task", {}).get("workflow") == "council-decision": + from .council_runtime import prepared_reports + artifacts = prepared_reports(store, run_id, snapshot, "cancelled") + version = store.cancel_council_queued(run_id, run["version"], artifacts) + return {"run_id": run_id, "state": "cancelled", "version": version} if run["state"] == "queued" and run["phase"] == "preparing": store.cancel_run_decision_observations(run_id) version = store.cancel_preparing(run_id) @@ -1456,7 +1508,7 @@ def cancel(self, run_id: str) -> dict[str, Any]: snapshot = self._review_snapshot(run) handoff = store.handoff_snapshot(run_id) if (snapshot.get("task", {}).get("workflow") - in {"branch-review", "issue-delivery"} + in {"branch-review", "issue-delivery", "council-decision"} and handoff is not None): terminal_artifacts = self._paused_review_terminal_artifacts( store, @@ -1501,7 +1553,7 @@ def handoff_claim( handoff_before_claim = store.handoff_snapshot(run_id) if (handoff_before_claim is not None and handoff_before_claim.packet.get("workflow") - in {"branch-review", "issue-delivery"} + in {"branch-review", "issue-delivery", "council-decision"} and self._review_snapshot(run)["task"]["lead"]["mode"] == "headless"): raise ConflictError( @@ -1930,6 +1982,69 @@ def _continue_headless_lead( "launch": launch, } + def council_handoff_view(self, run_id: str) -> dict[str, Any]: + """Validated portable Council evidence, without claiming or choosing.""" + from .council_runtime import validate_saved_handoff, verified_artifact + from .reports import handoff_report_names + store = self._store() + try: + run = store.run(run_id) + snapshot = self._review_snapshot(run) + handoff = store.handoff_snapshot(run_id) + if (handoff is None or run["state"] != "awaiting_host" or run["phase"] is not None + or handoff.status != "open" + or snapshot["task"].get("workflow") != "council-decision"): + raise ConflictError("Council host handoff is not open") + validate_saved_handoff(store, run_id, handoff, snapshot) + _, markdown = handoff_report_names(handoff.sequence) + report = verified_artifact(store, run_id, markdown).decode("utf-8") + report_artifact = store.artifact_named(run_id, markdown) + return {"run_id": run_id, "workflow": "council-decision", "version": run["version"], "packet_sha256": handoff.packet_sha256, + "candidate_sha256": handoff.packet["candidate_sha256"], "proposals": handoff.packet["proposals"], + "critique": handoff.packet["critique"], "checks": handoff.packet["checks"], "report": report, + "report_artifact": report_artifact, + "pending_finish": store.terminal_finish_decision(run_id, handoff.handoff_id)} + finally: + store.close() + + def finish_council(self, run_id: str, disposition: str, reason: str, *, chosen: str, + supported_claims: list[str], discarded_alternatives: list[str], validation: str) -> dict[str, Any]: + """Explicit guided Council choice over the existing exact host claim.""" + from .council import validate_choice + from .council_runtime import validate_saved_handoff, decision_gate + if not isinstance(reason, str) or not reason.strip() or len(reason) > 2000: + raise ContractError("finish requires a non-empty reason of at most 2000 characters") + reason = reason.strip() + store = self._store() + try: + run = store.run(run_id) + snapshot = self._review_snapshot(run) + handoff = store.handoff_snapshot(run_id) + if (handoff is None or run["state"] != "awaiting_host" or run["phase"] is not None + or handoff.status != "open" + or snapshot["task"].get("workflow") != "council-decision" or snapshot["task"]["lead"]["mode"] != "host"): + raise ConflictError("guided Council finish requires a current open host handoff") + validate_saved_handoff(store, run_id, handoff, snapshot) + choice = {"disposition": disposition, "reason": reason, "chosen": chosen, + "supported_claims": supported_claims, "discarded_alternatives": discarded_alternatives, + "validation": validation, "unresolved_objections": [item["id"] for item in handoff.packet["critique"]["objections"]]} + validate_choice(choice, snapshot["task"]["council"], handoff.packet["critique"]) + if disposition == "revise": + raise ContractError("Council additional rounds require a new explicitly capped run") + if disposition == "accept" and not handoff.packet["evaluation"]["accept_allowed"]: + raise ContractError("Council acceptance is blocked by mandatory checks/integrity") + body = {"schema_version": 1, "submission_id": "terminal-" + handoff.handoff_id, + "disposition": disposition, "reason": reason, "council_choice": choice, + "evidence_refs": [{"artifact_id": ref["artifact_id"], "sha256": ref["sha256"]} for ref in handoff.packet["artifacts"]]} + decision = {**body, "submission_hash": request_hash(body)} + decision_gate(store, run_id, handoff, snapshot, decision) + version = run["version"] + finally: + store.close() + claimed = self.handoff_claim(run_id, version, "terminal-operator", _initial_only=True, + _terminal_decision=decision) + return self.handoff_complete(run_id, claimed["claim"], decision) + def handoff_complete( self, run_id: str, @@ -1944,6 +2059,12 @@ def handoff_complete( try: run = store.run(run_id) handoff = store.handoff_snapshot_by_id(run_id, decoded.handoff_id) + if handoff.packet.get("workflow") == "council-decision": + from .council_runtime import complete + snapshot = self._review_snapshot(run) + if snapshot["task"]["lead"]["mode"] == "headless" and not decoded.owner_id.startswith("headless-lead:"): + raise ConflictError("headless Council does not accept a host completion") + return complete(self, store, run_id, handoff, snapshot, {**decision, "_claim": claim}) managed_review = handoff.packet.get("workflow") in { "branch-review", "issue-delivery", } @@ -2001,6 +2122,15 @@ def resume(self, run_id: str, recovery: dict[str, Any] | None = None) -> dict[st try: run = store.run(run_id) if run["state"] in TERMINAL_STATES: raise ConflictError("terminal run cannot resume; start a superseding run") + if run["state"] == "awaiting_host": + handoff = store.handoff_snapshot(run_id) + if handoff is not None and handoff.packet.get("workflow") == "council-decision": + from .council_runtime import continue_lead + continuation = continue_lead(self, store, run_id, run, handoff, self._review_snapshot(run)) + launch = continuation.pop("launch", None) + if launch is not None: + self._spawn_daemon(run_id, *launch) + return continuation if run["state"] == "awaiting_host" and run["phase"] is None: handoff = store.handoff_snapshot(run_id) if (handoff is None or handoff.packet.get("workflow") diff --git a/plugin/core/src/devsquad/store.py b/plugin/core/src/devsquad/store.py index 8fcec09..271018c 100644 --- a/plugin/core/src/devsquad/store.py +++ b/plugin/core/src/devsquad/store.py @@ -18,7 +18,7 @@ from .contracts import BudgetExhausted, ContractError -SUPPORTED_SCHEMA_VERSION = 16 +SUPPORTED_SCHEMA_VERSION = 17 TERMINAL_STATES = {"succeeded", "failed", "cancelled"} HOST_LEASE_SECONDS = 10 * 60 BRANCH_REVIEW_TERMINAL_ARTIFACTS = frozenset({ @@ -3017,7 +3017,7 @@ def reserve_attempt( ) -> AttemptReservation: if not owner_id or not package_digest: raise ContractError("supervisor owner and package digest are required") - if role not in {"worker", "implementer", "reviewer", "lead", "researcher"}: + if role not in {"worker", "implementer", "reviewer", "lead", "researcher", "proposer_a", "proposer_b", "critic"}: raise ContractError("attempt role is invalid") self.connection.execute("BEGIN IMMEDIATE") try: @@ -3630,6 +3630,7 @@ def commit_durable_handoff( artifacts: list[dict[str, Any]], metadata: Any, packet: dict[str, Any], + *, mutable_snapshot: dict[str, Any] | None = None, ) -> str: """Atomically import one completed attempt and publish its host packet.""" prepared, stdout_name, stderr_name = self._prepare_durable_artifacts( @@ -3669,6 +3670,24 @@ def commit_durable_handoff( if (attempt["status"] != "running" or run["state"] != "running" or run["phase"] is not None): raise ConflictError("durable handoff import is fenced") + if mutable_snapshot is not None: + from .council_runtime import verify_origin + current_snapshot = json.loads(self.run(run_id)["mutable_snapshot"]) + if (current_snapshot.get("task", {}).get("workflow") != "council-decision" + or attempt["id"] is None + or self.connection.execute("SELECT role FROM attempts WHERE id=?", (attempt["id"],)).fetchone()[0] != "critic"): + raise ConflictError("Council handoff projection is fenced") + verify_origin(self, run_id, current_snapshot) + verify_origin(self, run_id, mutable_snapshot) + old = current_snapshot["council_state"] + new = mutable_snapshot["council_state"] + if (old["next_role"] != "critic" or set(old["documents"]) != {"proposer_a", "proposer_b"} + or set(new["documents"]) != {"proposer_a", "proposer_b", "critic"} + or new["next_role"] != "lead" + or {key: value for key, value in new["documents"].items() if key != "critic"} != old["documents"] + or {key: value for key, value in new["artifacts"].items() if key != "critic"} != old["artifacts"]): + raise ConflictError("Council critic import changed finalized proposals or skipped the barrier") + self.connection.execute("UPDATE runs SET mutable_snapshot=? WHERE id=?", (canonical_json(mutable_snapshot), run_id)) version, artifact_ids = self._reference_prepared_artifacts( run_id, run["version"], prepared, @@ -3756,6 +3775,46 @@ def commit_durable_handoff( self.connection.execute("ROLLBACK") raise + def commit_council_stage(self, run_id: str, attempt_token: str, artifacts: list, + metadata: dict, snapshot: dict, role: str) -> str: + """Import one sealed proposal under the existing attempt fence.""" + from .council_runtime import verify_origin + prepared, stdout_name, stderr_name = self._prepare_durable_artifacts(run_id, artifacts, require_result_receipt=False) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.run(run_id) + attempt = self.connection.execute("SELECT * FROM attempts WHERE run_id=? AND attempt_token=?", (run_id, attempt_token)).fetchone() + if (not attempt or run["state"] != "running" or run["phase"] is not None + or attempt["status"] != "running" or attempt["role"] != role + or role not in {"proposer_a", "proposer_b"}): + raise ConflictError("Council proposal import is fenced") + prior = json.loads(run["mutable_snapshot"]) + verify_origin(self, run_id, prior) + verify_origin(self, run_id, snapshot) + if prior["council_state"]["next_role"] != role or role in prior["council_state"]["documents"]: + raise ConflictError("Council proposal stage is stale") + old_docs = prior["council_state"]["documents"] + new_docs = snapshot["council_state"]["documents"] + if (set(new_docs) != set(old_docs) | {role} + or {k: v for k, v in new_docs.items() if k != role} != old_docs + or snapshot["council_state"]["next_role"] != ("proposer_b" if role == "proposer_a" else "critic") + or {key: value for key, value in snapshot["council_state"]["artifacts"].items() if key != role} != prior["council_state"].get("artifacts", {})): + raise ConflictError("Council proposal import changed another finalized proposal") + version, ids = self._reference_prepared_artifacts(run_id, run["version"], prepared) + version = self._record_prepared_output(run_id, version, attempt, ids, stdout_name, stderr_name, canonical_json(metadata)) + now, version = _utc_now(), version + 1 + self.connection.execute("UPDATE attempts SET status='finished',finished_at=? WHERE id=?", (now, attempt["id"])) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("UPDATE runs SET mutable_snapshot=?,state='queued',phase=NULL,version=?,updated_at=? WHERE id=?", + (canonical_json(snapshot), version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'council.proposal_finalized',?,?)", + (run_id, version, canonical_json({"role": role, "attempt_id": attempt["id"]}), now)) + self.connection.execute("COMMIT") + return "queued" + except Exception: + self.connection.execute("ROLLBACK") + raise + def queue_headless_lead( self, run_id: str, @@ -3931,7 +3990,7 @@ def block_recovery(self, run_id: str, attempt_token: str, reason: str, *, releas @_project_terminal def recover_unstarted_attempt( - self, run_id: str, attempt_token: str, reason: str, + self, run_id: str, attempt_token: str, reason: str, *, terminal_artifacts: list | None = None, ) -> tuple[int, str]: """Recover a dead gated runner that never published a child identity. @@ -3939,6 +3998,8 @@ def recover_unstarted_attempt( atomic child record is absent. The inner gate cannot open before that record is durable, so no worker command can have executed in this case. """ + prepared = (self._prepare_exact_artifacts(run_id, terminal_artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS) + if terminal_artifacts is not None else None) self.connection.execute("BEGIN IMMEDIATE") try: run = self.connection.execute( @@ -3959,17 +4020,16 @@ def recover_unstarted_attempt( raise ConflictError("unstarted attempt now has a child identity record") now = _utc_now() if run["state"] == "cancelling": - path, digest, size, receipt_time = self._terminal_receipt( - run_id, - "cancelled", - "cancelling", - None, - attempt_id=attempt["id"], - now=now, - ) - version = self._reference_terminal_receipt( - run_id, run["version"], path, digest, size, receipt_time, - ) + 1 + if prepared is not None: + version, _ = self._reference_prepared_artifacts(run_id, run["version"], prepared) + version += 1 + else: + path, digest, size, receipt_time = self._terminal_receipt( + run_id, "cancelled", "cancelling", None, attempt_id=attempt["id"], now=now, + ) + version = self._reference_terminal_receipt( + run_id, run["version"], path, digest, size, receipt_time, + ) + 1 self.connection.execute( "UPDATE attempts SET status='finished',finished_at=? WHERE id=?", (now, attempt["id"]), @@ -4633,6 +4693,9 @@ def _validated_handoff_decision( "schema_version", "submission_id", "submission_hash", "disposition", "reason", "evidence_refs", } + snapshot = json.loads(self.run(run_id)["mutable_snapshot"] or "{}") + if snapshot.get("task", {}).get("workflow") == "council-decision": + expected_fields.add("council_choice") if not isinstance(decision, dict) or set(decision) != expected_fields: raise ContractError("handoff decision fields are invalid") if decision["schema_version"] != 1 or type(decision["schema_version"]) is not int: @@ -5393,6 +5456,32 @@ def fail_queued_budget( self.connection.execute("ROLLBACK") raise + @_project_terminal + def cancel_council_queued(self, run_id: str, expected_version: int, artifacts: list) -> int: + prepared = self._prepare_exact_artifacts(run_id, artifacts, BRANCH_REVIEW_TERMINAL_ARTIFACTS) + self.connection.execute("BEGIN IMMEDIATE") + try: + run = self.run(run_id) + snapshot = json.loads(run["mutable_snapshot"]) + if (run["version"] != expected_version or run["state"] != "queued" or run["phase"] is not None + or snapshot["task"]["workflow"] != "council-decision"): + raise ConflictError("Council queued cancellation is fenced") + if self.connection.execute("SELECT 1 FROM supervisor_claims WHERE run_id=? AND active=1", (run_id,)).fetchone(): + raise ConflictError("Council queued cancellation raced with an owned launcher") + version, _ = self._reference_prepared_artifacts(run_id, run["version"], prepared) + now, version = _utc_now(), version + 1 + self.connection.execute("UPDATE attempts SET status='recovery_required',finished_at=? WHERE run_id=? AND status='reserved'", (now, run_id)) + self.connection.execute("UPDATE supervisor_claims SET active=0 WHERE run_id=?", (run_id,)) + self.connection.execute("UPDATE claims SET active=0,fencing_token=fencing_token+1 WHERE run_id=?", (run_id,)) + self.connection.execute("UPDATE runs SET state='cancelled',phase=NULL,version=?,updated_at=? WHERE id=?", (version, now, run_id)) + self.connection.execute("INSERT INTO events(run_id,run_version,type,payload,created_at) VALUES(?,?,'run.cancelled',?,?)", + (run_id, version, canonical_json({"receipt": "result-receipt.json", "workflow": "council-decision"}), now)) + self.connection.execute("COMMIT") + return version + except Exception: + self.connection.execute("ROLLBACK") + raise + @_project_terminal def cancel_queued(self, run_id: str) -> int: self.connection.execute("BEGIN IMMEDIATE") diff --git a/plugin/core/src/devsquad/supervisor.py b/plugin/core/src/devsquad/supervisor.py index 08c7710..5534c27 100644 --- a/plugin/core/src/devsquad/supervisor.py +++ b/plugin/core/src/devsquad/supervisor.py @@ -326,7 +326,7 @@ def launch_durable( if not identity_committed: self.store.fail_launch(reservation,"durable gated launch failed") elif not gate_released: - self.store.recover_unstarted_attempt( + self._recover_unstarted_attempt( run_id, reservation.attempt_token, "runner gate could not be released", ) @@ -352,6 +352,15 @@ def wait_durable(self, handle: DurableAttempt, timeout_seconds: float) -> int: self.import_durable(handle.reservation.run_id) return returncode + def _recover_unstarted_attempt(self, run_id: str, attempt_token: str, reason: str) -> tuple[int, str]: + artifacts = None + run = self.store.run(run_id) + snapshot = json.loads(run["mutable_snapshot"]) + if run["state"] == "cancelling" and snapshot.get("task", {}).get("workflow") == "council-decision": + from .council_runtime import prepared_reports + artifacts = prepared_reports(self.store, run_id, snapshot, "cancelled") + return self.store.recover_unstarted_attempt(run_id, attempt_token, reason, terminal_artifacts=artifacts) + def _commit_review_handoff( self, run_id: str, @@ -579,7 +588,7 @@ def import_durable(self, run_id: str) -> str: self.store.block_recovery(run_id,attempt["attempt_token"],"runner and child died without an exit receipt",release_writer=True) return "recovery_required" try: - _, disposition = self.store.recover_unstarted_attempt( + _, disposition = self._recover_unstarted_attempt( run_id, attempt["attempt_token"], "runner died before publishing the gated child identity", @@ -618,6 +627,7 @@ def import_durable(self, run_id: str) -> str: managed_workflow = "internal_fake_delay" not in snapshot workflow_review = managed_workflow and workflow == "branch-review" workflow_delivery = managed_workflow and workflow == "issue-delivery" + workflow_council = managed_workflow and workflow == "council-decision" role = attempt.get("role", "worker") semantic_error=None native_failure = None @@ -634,6 +644,14 @@ def import_durable(self, run_id: str) -> str: ) except (ContractError, KeyError, TypeError, IndexError): pass + if (workflow_council and not receipt["cancelled"] and not receipt["timed_out"] and receipt["returncode"] == 0): + try: + from .council_runtime import import_stage + return import_stage(self.store, run_id, attempt, artifacts, metadata, snapshot, captures["stdout"]) + except ContractError as exc: + semantic_error = str(exc) + receipt["error"] = "COUNCIL_OUTPUT_INVALID" + receipt["message"] = semantic_error if (role == "implementer" and workflow_delivery and not receipt["cancelled"] and not receipt["timed_out"] and receipt["returncode"] == 0): @@ -701,9 +719,9 @@ def import_durable(self, run_id: str) -> str: if role == "lead" else "WORKFLOW_OUTPUT_INVALID" ) payload["message"]=semantic_error - if workflow_review or workflow_delivery: + if workflow_review or workflow_delivery or workflow_council: from .service import Service - prior_attempts, prior_dispositions = Service._saved_review_progress( + prior_attempts, prior_dispositions = ([], []) if workflow_council else Service._saved_review_progress( self.store, run_id, snapshot, @@ -788,6 +806,10 @@ def import_durable(self, run_id: str) -> str: metadata, fallback_error, ) + if workflow_council: + from .council_runtime import prepared_reports + artifacts.extend(prepared_reports(self.store, run_id, snapshot, terminal, error=report_error)) + return self.store.commit_durable_import(run_id, attempt["attempt_token"], artifacts, metadata, terminal, payload) reports = build_early_terminal_reports( run_id=run_id, state=terminal, diff --git a/plugin/core/src/devsquad/task_entry.py b/plugin/core/src/devsquad/task_entry.py index 55cbccd..f39a4b1 100644 --- a/plugin/core/src/devsquad/task_entry.py +++ b/plugin/core/src/devsquad/task_entry.py @@ -79,7 +79,8 @@ def discover_codex_identity( requested_effort: str | None = None, timeout_seconds: int = 15, runtime: Path | None = None, -) -> dict[str, str]: + all_models: bool = False, +) -> dict[str, str] | list[dict[str, str]]: """Discover one currently available exact Codex model without generating.""" manifest = AdapterManifest.load(CORE_ROOT / "adapters/codex/adapter.json") @@ -164,6 +165,18 @@ def read_native(request_id, method, params=None): if not candidates: qualifier = requested_model or "any model with effort metadata" raise ContractError(f"Codex model is unavailable: {qualifier}") + if all_models: + identities = [] + for candidate in sorted(candidates, key=lambda item: not item["is_default"]): + efforts = candidate["supported_efforts"] + if requested_effort is not None and requested_effort not in efforts: + continue + family = candidate.get("family") + identities.append({"harness": "codex", "harness_version": version, + "model_id": candidate["id"], "model_family": family if isinstance(family, str) and family else "gpt", + "effort": requested_effort or next((value for value in ("low", "medium") if value in efforts), efforts[0]), + **({"account_pool_id": pool_id, "native_scope": scope, "catalog_fingerprint": candidate["fingerprint"]} if scope is not None else {})}) + return identities selected = next( (model for model in candidates if model["is_default"]), candidates[0], diff --git a/plugin/core/src/devsquad/validation.py b/plugin/core/src/devsquad/validation.py index c0496e8..0b3554f 100644 --- a/plugin/core/src/devsquad/validation.py +++ b/plugin/core/src/devsquad/validation.py @@ -5,7 +5,7 @@ from typing import Any from .contracts import ContractError -TASK_FIELDS = {"schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin", "review"} +TASK_FIELDS = {"schema_version", "project", "workflow", "goal", "task_class", "acceptance", "checks", "scope", "lead", "routing", "budget", "origin", "review", "council"} MAX_ACCEPTANCE_CRITERIA = 100 MAX_CHECKS = 16 MAX_SCOPE_PATHS = 256 @@ -28,9 +28,14 @@ def _relative(path: str, label: str) -> None: def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) -> None: - _exact(value, TASK_FIELDS, TASK_FIELDS - {"review"}, "task") - if type(value["schema_version"]) is not int or value["schema_version"] != 1 or not isinstance(value["workflow"], str) or value["workflow"] not in {"branch-review", "issue-delivery"}: + _exact(value, TASK_FIELDS, TASK_FIELDS - {"review", "council"}, "task") + if type(value["schema_version"]) is not int or value["schema_version"] != 1 or not isinstance(value["workflow"], str) or value["workflow"] not in {"branch-review", "issue-delivery", "council-decision"}: raise ContractError("unsupported task schema or workflow") + if value["workflow"] == "council-decision": + from .council import validate_spec + validate_spec(value.get("council")) + elif "council" in value: + raise ContractError("CouncilSpec requires the council-decision workflow") project = value["project"] _exact(project, {"repo_path", "base_ref", "target_ref"}, {"repo_path", "base_ref", "target_ref"}, "project") if not all(isinstance(project[k], str) and project[k] for k in ("repo_path", "base_ref", "target_ref")) or not Path(project["repo_path"]).is_absolute() or (require_existing_repo and not (Path(project["repo_path"]) / ".git").exists()): @@ -76,7 +81,7 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) if len(scope["read_paths"]) + len(scope["write_paths"]) > MAX_SCOPE_PATHS: raise ContractError("scope paths exceed their bound") if len(set(scope["read_paths"])) != len(scope["read_paths"]) or len(set(scope["write_paths"])) != len(scope["write_paths"]): raise ContractError("scope paths must be unique") for p in scope["read_paths"] + scope["write_paths"]: _relative(p, "scope path") - if value["workflow"] == "branch-review" and scope["write_paths"]: raise ContractError("branch review cannot write") + if value["workflow"] in {"branch-review", "council-decision"} and scope["write_paths"]: raise ContractError("read-only workflow cannot write") lead = value["lead"]; _exact(lead, {"mode"}, {"mode"}, "lead") if not isinstance(lead["mode"], str) or lead["mode"] not in {"host", "headless"}: raise ContractError("invalid lead mode") routing = value["routing"] @@ -102,7 +107,7 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) overrides = routing.get("overrides", {}) if not isinstance(overrides, dict): raise ContractError("routing overrides must be an object") for role, override in overrides.items(): - if role not in {"implementer", "reviewer", "lead", "researcher"}: raise ContractError("invalid override role") + if role not in {"implementer", "reviewer", "lead", "researcher", "proposer_a", "proposer_b", "critic"}: raise ContractError("invalid override role") _exact(override, {"profile_id", "fallback"}, {"profile_id"}, "routing override") if not isinstance(override["profile_id"], str) or not override["profile_id"]: raise ContractError("override profile_id must be non-empty") if not isinstance(override.get("fallback", "none"), str) or override.get("fallback", "none") not in {"none", "policy"}: raise ContractError("override fallback must be none or policy") @@ -118,6 +123,14 @@ def validate_task(value: dict[str, Any], *, require_existing_repo: bool = False) for key, number in budget.items(): if not isinstance(number, int) or isinstance(number, bool) or number < 0: raise ContractError(f"budget {key} must be a finite non-negative integer") if budget["wall_seconds"] == 0 or budget["max_worker_invocations"] == 0: raise ContractError("wall_seconds and max_worker_invocations must be positive") + if value["workflow"] == "council-decision": + minimum = 4 if lead["mode"] == "headless" else 3 + if budget["max_worker_invocations"] < minimum or budget["max_worker_invocations"] > value["council"]["max_invocations"]: + raise ContractError("Council worker budget must fit the explicit Council invocation cap") + if budget["max_revisions"] != 0: + raise ContractError("bounded Council supports one round; additional rounds need a new capped run") + if "review" in value: + raise ContractError("Council uses its frozen rubric, not branch-review options") def validate_profile(value: dict[str, Any]) -> None: @@ -171,7 +184,7 @@ def validate_policy(value: dict[str, Any]) -> None: if type(value["schema_version"]) is not int or value["schema_version"] != 1 or type(value["version"]) is not int or value["version"] < 1: raise ContractError("invalid policy version") if not isinstance(value["id"], str) or not value["id"]: raise ContractError("policy id must be non-empty") if type(value["require_different_model_for_review"]) is not bool or ("prefer_different_harness_for_review" in value and type(value["prefer_different_harness_for_review"]) is not bool): raise ContractError("policy review flags must be boolean") - if not isinstance(value["roles"], dict) or set(value["roles"]) - {"implementer", "reviewer", "lead", "researcher"}: raise ContractError("invalid policy roles") + if not isinstance(value["roles"], dict) or set(value["roles"]) - {"implementer", "reviewer", "lead", "researcher", "proposer_a", "proposer_b", "critic"}: raise ContractError("invalid policy roles") for candidates in value["roles"].values(): if not isinstance(candidates, list) or not candidates: raise ContractError("role candidates must be non-empty arrays") for ref in candidates: diff --git a/test/core/test_capacity.py b/test/core/test_capacity.py index e097cac..a91b5a6 100644 --- a/test/core/test_capacity.py +++ b/test/core/test_capacity.py @@ -15,7 +15,7 @@ from devsquad.capacity import derive_pool_capacity, validate_observation from devsquad.contracts import ContractError -from devsquad.store import ConflictError, SchemaVersionError, Store +from devsquad.store import ConflictError, SchemaVersionError, Store, SUPPORTED_SCHEMA_VERSION NOW = datetime(2026, 9, 27, 15, 0, tzinfo=timezone.utc) @@ -185,7 +185,7 @@ def test_migration_nine_creates_capacity_ledger(self): version = store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 16) + self.assertEqual(version, SUPPORTED_SCHEMA_VERSION) tables = { row[0] for row in store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_council_comparison.py b/test/core/test_council_comparison.py new file mode 100644 index 0000000..6596071 --- /dev/null +++ b/test/core/test_council_comparison.py @@ -0,0 +1,92 @@ +from __future__ import annotations + +import copy +import json +from pathlib import Path +import sys +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.council_comparison import predeclare, report +from devsquad.contracts import ContractError +from devsquad.store import Store, request_hash, ConflictError +import test_council_runtime as council_fixture_tests + + +class CouncilComparisonTest(unittest.TestCase): + setUp = council_fixture_tests.CouncilRuntimeTest.setUp + git = council_fixture_tests.CouncilRuntimeTest.git + wait = council_fixture_tests.CouncilRuntimeTest.wait + receipt = council_fixture_tests.CouncilRuntimeTest.receipt + + def test_predeclared_actual_matched_heldout_process_receipts_are_inconclusive_not_native_quality(self): + profiles = json.loads((self.repo / "profiles.json").read_text()) + policy = json.loads((self.repo / "policy.json").read_text()) + cases = [] + for identifier, split, question in (("retry-key", "matched", "Compare retry key retention alternatives"), + ("retry-expiry", "heldout", "Compare safe retry after key expiry")): + council = copy.deepcopy(self.task) + council["goal"] = question + council["project"]["base_ref"] = self.source_oid + council["project"]["target_ref"] = self.source_oid + council["lead"]["mode"] = "host" + council["routing"] = {"profiles": profiles, "policy": policy} + control = copy.deepcopy(council) + control["workflow"] = "branch-review" + control.pop("council") + control["routing"]["policy"]["roles"] = {"reviewer": [{"kind": "profile", "id": "critic"}]} + cases.append({"id": identifier, "split": split, "control": control, "council": council}) + declaration = predeclare(self.root / "predeclared-workflow-comparison.json", cases) + copied = copy.deepcopy(cases[0]) + copied["id"], copied["split"] = "renamed-repetition", "heldout" + with self.assertRaises(ContractError): + predeclare(self.root / "invalid-heldout.json", [cases[0], copied]) + with self.assertRaises(FileExistsError): + predeclare(Path(declaration["path"]), cases) + pairs = [] + self.task["lead"]["mode"] = "host" # wait helper does not auto-submit + for case in cases: + control = self.service.start(case["control"], case["id"] + "-control", + _internal_review_fixture={"verdict": "clean", "summary": "Controlled public review", "findings": []}) + waiting = self.wait(control["run_id"], {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + claimed = self.service.handoff_claim(control["run_id"], waiting["version"], "public-fixture-host") + body = {"schema_version": 1, "submission_id": "public-control", "disposition": "accept", "reason": "Controlled mechanical acceptance", + "evidence_refs": [{"artifact_id": item["artifact_id"], "sha256": item["sha256"]} for item in claimed["handoff"]["packet"]["artifacts"]]} + self.service.handoff_complete(control["run_id"], claimed["claim"], {**body, "submission_hash": request_hash(body)}) + council = self.service.start(case["council"], case["id"] + "-council", _internal_council_fixture=self.fixture) + self.assertEqual(self.wait(council["run_id"], {"awaiting_host", "failed"})["state"], "awaiting_host") + self.service.finish_council(council["run_id"], "accept", "Controlled mechanical acceptance", chosen="synthesis", + supported_claims=["No blind retry"], discarded_alternatives=["Blind retry"], validation="Test duplicate effects") + pairs.append({"id": case["id"], "control": control["run_id"], "council": council["run_id"]}) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + self.addCleanup(store.close) + compared = report(store, declaration, pairs) + self.assertTrue(Path(compared["path"]).is_file()) + self.assertEqual(report(store, declaration, pairs), compared) + self.assertEqual(compared["report"]["conclusion"], "inconclusive") + self.assertFalse(compared["report"]["automatic_enabled"]) + self.assertEqual({case["split"] for case in compared["report"]["cases"]}, {"matched", "heldout"}) + for case in compared["report"]["cases"]: + self.assertEqual(case["arms"]["control"]["worker_invocations"], 1) + self.assertEqual(case["arms"]["council"]["worker_invocations"], 3) + for arm in case["arms"].values(): + self.assertIsNone(arm["accepted_quality"]) + self.assertIsNone(arm["escaped_defects"]) + self.assertIsNone(arm["rework"]) + self.assertGreaterEqual(arm["execution_elapsed_ms"], 0) + self.assertTrue(all(arm["actual_prompt_sha256"])) + with self.assertRaises(ContractError): + report(store, declaration, [pairs[0], pairs[0]]) + original = Path(declaration["path"]).read_bytes() + tampered = json.loads(original) + tampered["automatic_enabled"] = True + Path(declaration["path"]).write_text(json.dumps(tampered)) + with self.assertRaises(ConflictError): + report(store, declaration, pairs) + Path(declaration["path"]).write_bytes(original) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_contract.py b/test/core/test_council_contract.py new file mode 100644 index 0000000..726824f --- /dev/null +++ b/test/core/test_council_contract.py @@ -0,0 +1,79 @@ +from __future__ import annotations + +import unittest +import copy +from pathlib import Path +import sys + +sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "plugin/core/src")) + +from devsquad.contracts import ContractError +from devsquad.council import digest, label_mapping, validate_spec, validate_critique, PROMPT_VERSION, output_schema, prompt_digest +from devsquad.council_runtime import validate_document + + +class CouncilContractTest(unittest.TestCase): + def spec(self): + return {"schema_version": 1, "enabled": True, "automatic": False, + "reason": "Resolve the competing hypotheses", "min_valid_proposals": 2, + "required_critics": 1, "max_invocations": 4, "seed": "a" * 64, + "evidence": [], "rubric": [{"id": "correctness", "description": "Fits evidence"}]} + + def test_strict_spec_and_automatic_off(self): + validate_spec(self.spec()) + for changes in ({"automatic": True}, {"min_valid_proposals": 1}, + {"required_critics": 0}, {"unknown": 1}): + with self.assertRaises(ContractError): + validate_spec({**self.spec(), **changes}) + + def test_seeded_labels_are_reversible(self): + mapping = label_mapping(self.spec()["seed"], ["proposer_a", "proposer_b"]) + self.assertEqual(set(mapping), {"A", "B"}) + self.assertEqual(set(mapping.values()), {"proposer_a", "proposer_b"}) + self.assertEqual(mapping, label_mapping(self.spec()["seed"], ["proposer_a", "proposer_b"])) + + def test_missing_unknown_duplicate_criterion_or_label_is_invalid(self): + good = {"summary": "Both considered", "assessments": [ + {"label": label, "criterion_id": "correctness", "status": "supported", + "reason": "Matches the packet", "evidence_ids": []} for label in ("A", "B")], + "objections": [{"id": "o1", "label": "B", "reason": "Unmeasured alternative", "evidence_ids": []}]} + validate_critique(good, self.spec()) + for assessments in (good["assessments"][:1], good["assessments"] * 2, + [{**good["assessments"][0], "label": "C"}, good["assessments"][1]]): + with self.assertRaises(ContractError): + validate_critique({**good, "assessments": assessments}, self.spec()) + + def test_native_import_requires_exact_frozen_identity_correlated_ids_and_usage(self): + profile = {"id": "author", "model_id": "actual-model", "effort": {"value": "high"}} + selected = {"profile_id": "author", "profile": profile, "profile_sha256": digest(profile)} + adapter = {"harness": "codex", "harness_version": "codex-cli tested", "model_provider": "openai"} + brief = {"source_files": []} + snapshot = {"routing": {"roles": {"proposer_a": {"selected": selected, "fallbacks": []}}}, + "council_brief": brief, "council_fixture": None, "task": {"council": self.spec()}, + "council_state": {"documents": {}}, + "council_adapters": {"proposer_a": {"author": adapter}}, + "council_boundaries": {"proposer_a": {"author": {"profile_sha256": "b" * 64}}}} + evidence = {"schema_version": 1, "role": "proposer_a", "profile": profile, "profile_sha256": digest(profile), + "prompt_version": PROMPT_VERSION, "prompt_sha256": prompt_digest("proposer_a", {"brief": brief}), + "role_packet_sha256": digest({"brief": brief}), "output_schema_sha256": digest(output_schema("proposer_a")), + "brief_sha256": digest(brief), "identity_scope": "native_verified", "boundary_sha256": "b" * 64, + "observed_identity": {**adapter, "model_id": "actual-model", "effort": "high", "permission_policy": "read_only", "verification": "verified"}, + "native_ids": {"thread_id": "observed-thread", "turn_id": "observed-turn"}, + "usage": {"input_tokens": None, "output_tokens": None, "total_tokens": None, "source": "unavailable"}, + "document": {"summary": "Proposal", "approach": "Approach", "claims": [{"text": "Claim", "evidence_ids": []}], "validation": "Check"}, "checks": []} + attempt = {"role": "proposer_a", "profile_id": "author", "profile_index": 0} + validate_document(evidence, snapshot, attempt) + invalid = [] + for key, value in (("effort", "low"), ("harness_version", "other"), ("model_provider", "other"), ("permission_policy", "write"), ("model_id", "requested-only")): + altered = copy.deepcopy(evidence) + altered["observed_identity"][key] = value + invalid.append(altered) + for field, value in (("native_ids", {"thread_id": "thread"}), ("native_ids", {"thread_id": "thread", "turn_id": ""}), + ("usage", {"input_tokens": 0, "output_tokens": 0, "total_tokens": 0, "source": "unavailable"}), + ("observed_identity", {"verification": "verified", "model_id": "actual-model"})): + altered = copy.deepcopy(evidence) + altered[field] = value + invalid.append(altered) + for altered in invalid: + with self.assertRaises(ContractError): + validate_document(altered, snapshot, attempt) diff --git a/test/core/test_council_entry.py b/test/core/test_council_entry.py new file mode 100644 index 0000000..e23eb39 --- /dev/null +++ b/test/core/test_council_entry.py @@ -0,0 +1,84 @@ +from __future__ import annotations + +from pathlib import Path +import contextlib +import io +import json +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.council_task_entry import build_council_task +from devsquad.contracts import CapabilityUnavailable, ContractError +from devsquad import cli + + +class CouncilEntryTest(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="council-entry-public-") + self.addCleanup(self.temporary.cleanup) + self.repo = Path(self.temporary.name) + for args in (("init", "-q"), ("config", "user.email", "fixture@example.invalid"), + ("config", "user.name", "Public fixture")): + self.git(*args) + (self.repo / "README").write_text("Frozen public input\n") + self.git("add", "README") + self.git("commit", "-qm", "Frozen public input") + self.identities = [{"harness": "codex", "model_id": "model-" + name, "model_family": name, + "effort": "low", "harness_version": "codex-cli test", "account_pool_id": "subscription-shared"} + for name in ("a", "b", "c")] + + def git(self, *args): + return subprocess.run(["git", "-C", str(self.repo), *args], check=True, capture_output=True, text=True).stdout.strip() + + def test_manual_scope_budget_and_selection_are_explicit_without_launch(self): + task, summary = build_council_task(project_dir=self.repo, goal="Compare alternatives", identities=self.identities, + read_paths=["README"], max_invocations=4) + self.assertEqual(task["scope"], {"read_paths": ["README"], "write_paths": []}) + self.assertEqual(task["budget"]["max_worker_invocations"], 4) + self.assertEqual(task["budget"]["max_revisions"], 0) + self.assertFalse(summary["automatic_enabled"]) + self.assertEqual(set(summary["profiles"]), {"proposer_a", "proposer_b", "critic"}) + self.assertTrue(all(profile["quality_status"] == "trial" for profile in summary["profiles"].values())) + self.assertEqual(self.git("status", "--porcelain"), "") + + def test_catalog_duplicates_missing_identity_and_unsafe_scope_fail_closed(self): + with self.assertRaises(CapabilityUnavailable): + build_council_task(project_dir=self.repo, goal="Question", identities=self.identities[:2] + [self.identities[0]]) + incomplete = [{**identity} for identity in self.identities] + incomplete[1].pop("harness_version") + with self.assertRaises(ContractError): + build_council_task(project_dir=self.repo, goal="Question", identities=incomplete) + for path in ("../peer", "/private/peer"): + with self.assertRaises(ContractError): + build_council_task(project_dir=self.repo, goal="Question", identities=self.identities, read_paths=[path]) + + def test_normal_dry_run_human_and_json_preserve_explicit_scope_and_cap(self): + for json_mode in (False, True): + output = io.StringIO() + with patch.object(cli, "discover_codex_identity", return_value=self.identities), contextlib.redirect_stdout(output): + args = ["council", "Compare retry alternatives", "--project-dir", str(self.repo), + "--read-path", "README", "--lead", "host", "--max-invocations", "3", "--dry-run"] + if json_mode: + args.append("--json") + self.assertEqual(cli.main(args), 0) + if json_mode: + value = json.loads(output.getvalue()) + self.assertEqual(set(value), {"schema_version", "ok", "data", "error"}) + self.assertTrue(value["data"]["dry_run"]) + self.assertIsNone(value["data"]["run_id"]) + self.assertFalse(value["data"]["native_ready"]) + self.assertEqual(value["data"]["scope"]["read_paths"], ["README"]) + else: + self.assertIn("proposer_a: codex model-a", output.getvalue()) + self.assertIn("Read scope: README; write scope: none", output.getvalue()) + self.assertIn("Worker cap: 3; rounds: 1; lead: host; automatic off", output.getvalue()) + self.assertIn("Native readiness: unavailable", output.getvalue()) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_epoch.py b/test/core/test_council_epoch.py new file mode 100644 index 0000000..af57363 --- /dev/null +++ b/test/core/test_council_epoch.py @@ -0,0 +1,58 @@ +"""Contract-epoch fences; immutable installed-old-client proof is separate.""" +from pathlib import Path +import sqlite3 +import subprocess +import sys +import tempfile +import types +import unittest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.store import Store, SchemaVersionError, SUPPORTED_SCHEMA_VERSION + + +class CouncilEpochTest(unittest.TestCase): + def test_schema16_active_and_recoverable_deferral_then_upgrade_and_old_connection_fence(self): + # A separately loaded module keeps its connection-schema function at 16 + # after current Store commits 17. This tests existing hot-client triggers, + # not the older Service's permission semantics or installed package proof. + old = types.ModuleType("devsquad._controlled_schema16_store") + old.__package__ = "devsquad" + old.__file__ = str(ROOT / "plugin/core/src/devsquad/store.py") + sys.modules[old.__name__] = old + self.addCleanup(sys.modules.pop, old.__name__, None) + source = Path(old.__file__).read_text().replace(f"SUPPORTED_SCHEMA_VERSION = {SUPPORTED_SCHEMA_VERSION}", "SUPPORTED_SCHEMA_VERSION = 16", 1) + exec(compile(source, old.__file__, "exec"), old.__dict__) + with tempfile.TemporaryDirectory(prefix="council-epoch-public-") as temporary: + root = Path(temporary) + repo = root / "repo" + subprocess.run(["git", "init", "-q", str(repo)], check=True) + database, artifacts = root / "state.sqlite3", root / "artifacts" + previous = old.Store(database, artifacts) + self.addCleanup(previous.close) + before_tables = {r[0] for r in previous.connection.execute("SELECT name FROM sqlite_master WHERE type='table'")} + active = previous.claim_start(repo, "active16", {"task": "legacy"}, "old16") + with self.assertRaisesRegex(SchemaVersionError, "deferred.*active/recoverable"): + Store(database, artifacts) + previous.complete_preparation(active.run_id, active.fencing_token, {"legacy": True}) + with self.assertRaisesRegex(SchemaVersionError, "deferred.*active/recoverable"): + Store(database, artifacts) + self.assertEqual(previous.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], 16) + previous.cancel_queued(active.run_id) + upgraded = Store(database, artifacts) + try: + self.assertEqual(upgraded.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0], SUPPORTED_SCHEMA_VERSION) + self.assertGreaterEqual(SUPPORTED_SCHEMA_VERSION, 17) + self.assertEqual(before_tables, {r[0] for r in upgraded.connection.execute("SELECT name FROM sqlite_master WHERE type='table'")}) + with self.assertRaisesRegex(sqlite3.DatabaseError, "newer than connection supports"): + previous.claim_start(repo, "old16-after17", {"task": "forbidden"}, "old16") + with self.assertRaisesRegex(old.SchemaVersionError, "newer than supported 16"): + old.Store(database, artifacts) + self.assertEqual(upgraded.connection.execute("SELECT COUNT(*) FROM runs").fetchone()[0], 1) + finally: + upgraded.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_install_epoch.py b/test/core/test_council_install_epoch.py new file mode 100644 index 0000000..ac22662 --- /dev/null +++ b/test/core/test_council_install_epoch.py @@ -0,0 +1,229 @@ +"""Actual temporary installer gate using exact accepted R5 core provenance.""" +from __future__ import annotations + +from contextlib import closing +import hashlib +import io +import json +from pathlib import Path +import sqlite3 +import subprocess +import tarfile +import time +import unittest + +import test_install_core as installer_tests +import test_council_runtime as council_fixture_tests +from devsquad.store import SUPPORTED_SCHEMA_VERSION + +ROOT = Path(__file__).resolve().parents[2] +OLD_COMMIT = "bf3de0867484552d354e6e8b6ba835f31a93aeb3" +OLD_SOURCE_SHA256 = "90f1e87fb9acf7756dad46780672de9b84fe75e3631322857a310f089280345e" +OLD_PACKAGE_SHA256 = "e03bf3a2e362b3aff6d0a5dd00bd837f15afb216d4c58ac1d9c776c4d8f09c67" + + +class CouncilInstallEpochTest(unittest.TestCase): + setUp = installer_tests.StandaloneInstallerTest.setUp + tearDown = installer_tests.StandaloneInstallerTest.tearDown + install = installer_tests.StandaloneInstallerTest.install + cli_json = installer_tests.StandaloneInstallerTest.cli_json + + def _schema(self, runtime): + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + return connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] + + def _wait(self, launcher, runtime, run_id, states): + deadline = time.monotonic() + 12 + while time.monotonic() < deadline: + status = self.cli_json(launcher, "status", run_id, "--runtime-dir", str(runtime))["data"] + if status["state"] in states: + return status + time.sleep(.05) + self.fail(f"Controlled installed run did not reach {states}: {status}") + + def _python(self, python, script, *arguments): + completed = subprocess.run([str(python), "-P", "-c", script, *map(str, arguments)], + capture_output=True, text=True, check=True, env=self.environment, cwd=self.root, timeout=20) + return json.loads(completed.stdout) + + def test_exact_old16_install_deferral_reconciliation_upgrade_and_old_handoff_mutation_fences(self): + # This archive's entire core digest is exactly the accepted immutable + # installed R5 payload, not a rewritten current Store with a lower label. + archived = subprocess.run(["git", "archive", OLD_COMMIT + ":plugin/core"], + cwd=ROOT, capture_output=True, check=True).stdout + old_source = self.root / "accepted-schema16-core" + old_source.mkdir() + with tarfile.open(fileobj=io.BytesIO(archived)) as archive: + digest = hashlib.sha256() + for member in sorted((member for member in archive.getmembers() if member.isfile()), key=lambda member: member.name): + digest.update(member.name.encode() + b"\0" + archive.extractfile(member).read()) + self.assertEqual(digest.hexdigest(), OLD_SOURCE_SHA256) + archive.extractall(old_source, filter="data") + self.environment["PIP_NO_INDEX"] = "1" + self.environment["DEVSQUAD_RUNTIME_DIR"] = str(self.install_root / "runtime") + accepted_python = Path("/Users/Dikshant/.devsquad/releases/0.1.0-py31214-90f1e87fb9ac-mcp-a26bc88afbef/venv/bin/python") + if accepted_python.is_file(): + self.environment["DEVSQUAD_PYTHON"] = str(accepted_python) + first = self.install(old_source) + old_release = Path(first["current_target"]) + old_python = old_release / "venv/bin/python" + self.assertEqual(json.loads((old_release / "release.json").read_text())["source_digest"], OLD_SOURCE_SHA256) + runtime, launcher = self.install_root / "runtime", self.bin_dir / "squad" + + # Build harmless public input/fixture documents; all execution below uses + # the actual temporary installed packages through their own interpreters. + fixture = council_fixture_tests.CouncilRuntimeTest() + fixture.setUp() + self.addCleanup(fixture.doCleanups) + profiles = json.loads((fixture.repo / "profiles.json").read_text()) + policy = json.loads((fixture.repo / "policy.json").read_text()) + control = json.loads(json.dumps(fixture.task)) + control.pop("council") + control["workflow"] = "branch-review" + control["lead"]["mode"] = "host" + control["routing"] = {"profiles": profiles, "policy": json.loads(json.dumps(policy))} + control["routing"]["policy"]["roles"] = {"reviewer": [{"kind": "profile", "id": "critic"}]} + task_path = self.root / "controlled-old16-task.json" + task_path.write_text(json.dumps(control)) + script = """ +import json,sys +from pathlib import Path +from unittest.mock import patch +from devsquad.service import Service +from devsquad.store import SUPPORTED_SCHEMA_VERSION +s=Service(Path(sys.argv[2])) +task=json.loads(Path(sys.argv[1]).read_text()) +live=s.start(task,'accepted16-active',_internal_fake_delay=30) +with patch.object(s,'_spawn_daemon',return_value=0): + queued=s.start(task,'accepted16-recoverable',_internal_fake_delay=.1) +print(json.dumps({'supported':SUPPORTED_SCHEMA_VERSION,'live':live,'queued':queued})) +""" + admitted = self._python(old_python, script, task_path, runtime) + self.assertEqual(admitted["supported"], 16) + live, queued = admitted["live"]["run_id"], admitted["queued"]["run_id"] + hot = None + council_id = None + try: + self.assertEqual(self._wait(launcher, runtime, live, {"running"})["state"], "running") + hot_script = """ +import json,sys +from pathlib import Path +from devsquad.store import Store,SUPPORTED_SCHEMA_VERSION +from devsquad.service import Service +runtime=Path(sys.argv[1]); s=Store(runtime/'state.sqlite3',runtime/'artifacts') +package=Service(runtime)._freeze_package()[1] +print(json.dumps({'supported':SUPPORTED_SCHEMA_VERSION,'schema':s.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0],'package_digest':package}),flush=True) +request=json.loads(sys.stdin.readline()) +try: + s.claim_handoff(request['run_id'],request['version'],'old16-long-lived-owner') +except Exception as exc: + print(json.dumps({'rejected':True,'error':str(exc)}),flush=True) +else: + print(json.dumps({'rejected':False}),flush=True) +finally: + s.close() +""" + hot = subprocess.Popen([str(old_python), "-P", "-c", hot_script, str(runtime)], + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + text=True, env=self.environment, cwd=self.root) + ready = json.loads(hot.stdout.readline()) + self.assertEqual(ready, {"supported": 16, "schema": 16, "package_digest": OLD_PACKAGE_SHA256}) + before_selector = (self.install_root / "current").resolve() + before_launcher = launcher.read_bytes() + before_state = (self.install_root / "install-state.json").read_bytes() + + def assert_deferred(): + attempted = subprocess.run(["/bin/bash", str(installer_tests.INSTALLER), "--source-core", str(installer_tests.CORE), "--json"], + capture_output=True, text=True, env=self.environment, cwd=ROOT, timeout=90) + self.assertNotEqual(attempted.returncode, 0) + self.assertIn("active/recoverable", attempted.stderr) + self.assertEqual((self.install_root / "current").resolve(), before_selector) + self.assertEqual(launcher.read_bytes(), before_launcher) + self.assertEqual((self.install_root / "install-state.json").read_bytes(), before_state) + self.assertEqual(self._schema(runtime), 16) + + assert_deferred() + self.cli_json(launcher, "cancel", live, "--runtime-dir", str(runtime)) + self.assertEqual(self._wait(launcher, runtime, live, {"cancelled"})["state"], "cancelled") + # Active work is gone, but a recoverable old run still defers update. + assert_deferred() + self.cli_json(launcher, "resume", queued, "--runtime-dir", str(runtime)) + self.assertEqual(self._wait(launcher, runtime, queued, {"succeeded"})["state"], "succeeded") + self.assertEqual(self._schema(runtime), 16) + updated = self.install() + new_release = Path(updated["current_target"]) + self.assertNotEqual(new_release, old_release) + self.assertTrue(old_release.is_dir()) + self.assertEqual(self._schema(runtime), 16) # selector activation does not migrate + self.assertTrue(self.cli_json(launcher, "result", queued, "--runtime-dir", str(runtime))["data"]["ready"]) + self.assertEqual(self._schema(runtime), SUPPORTED_SCHEMA_VERSION) + + council = json.loads(json.dumps(fixture.task)) + council["lead"]["mode"] = "host" + council["routing"] = {"profiles": profiles, "policy": policy} + payload = self.root / "controlled-council.json" + payload.write_text(json.dumps({"task": council, "fixture": fixture.fixture})) + start_script = """ +import json,sys +from pathlib import Path +from devsquad.service import Service +value=json.loads(Path(sys.argv[1]).read_text()) +print(json.dumps(Service(Path(sys.argv[2])).start(value['task'],'new17-host-council',_internal_council_fixture=value['fixture']))) +""" + council_id = self._python(new_release / "venv/bin/python", start_script, payload, runtime)["run_id"] + waiting = self._wait(launcher, runtime, council_id, {"awaiting_host", "failed"}) + self.assertEqual(waiting["state"], "awaiting_host") + output, stderr = hot.communicate(json.dumps({"run_id": council_id, "version": waiting["version"]}) + "\n", timeout=8) + self.assertEqual(hot.returncode, 0, stderr) + hot_result = json.loads(output) + self.assertTrue(hot_result["rejected"]) + self.assertIn("newer than connection supports", hot_result["error"]) + fresh_script = """ +import json,sys +from pathlib import Path +from devsquad.service import Service +try: + Service(Path(sys.argv[1])).handoff_claim(sys.argv[2],int(sys.argv[3]),'old16-fresh-owner') +except Exception as exc: + print(json.dumps({'rejected':True,'type':type(exc).__name__,'error':str(exc)})) +else: + print(json.dumps({'rejected':False})) +""" + fresh = self._python(old_python, fresh_script, runtime, council_id, waiting["version"]) + self.assertTrue(fresh["rejected"]) + self.assertEqual(fresh["type"], "SchemaVersionError") + self.assertIn(f"{SUPPORTED_SCHEMA_VERSION} is newer than supported 16", fresh["error"]) + after = self.cli_json(launcher, "status", council_id, "--runtime-dir", str(runtime))["data"] + self.assertEqual(after["version"], waiting["version"]) + self.assertIsNone(after["handoff"]["claimed_by"]) + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + self.assertEqual(connection.execute("SELECT COUNT(*) FROM claims WHERE run_id=? AND handoff_id IS NOT NULL AND active=1", (council_id,)).fetchone()[0], 0) + self.cli_json(launcher, "cancel", council_id, "--runtime-dir", str(runtime)) + self.assertEqual(self._wait(launcher, runtime, council_id, {"cancelled"})["state"], "cancelled") + from devsquad.supervisor import _live_group_exists + with closing(sqlite3.connect(runtime / "state.sqlite3")) as connection: + groups = [row[0] for row in connection.execute("SELECT pgid FROM attempts WHERE pgid IS NOT NULL")] + self.assertEqual(connection.execute("SELECT COUNT(*) FROM runs WHERE state NOT IN ('succeeded','failed','cancelled')").fetchone()[0], 0) + self.assertTrue(all(not _live_group_exists(group) for group in groups)) + self.epoch_probe_result = {"old_commit": OLD_COMMIT, "old_source_sha256": OLD_SOURCE_SHA256, + "old_package_sha256": OLD_PACKAGE_SHA256, "old_schema": 16, "new_schema": self._schema(runtime), + "active_deferral": True, "recoverable_deferral": True, "selector_and_launcher_preserved": True, + "old_active_cancelled": True, "old_recoverable_reconciled": True, + "long_lived_old_claim_rejected": True, "fresh_old_service_claim_rejected": True, + "council_claim_unchanged": True, "native_generation": False, "mcp_installed_in_temporary_release": False, + "owned_process_groups_gone": True, + "candidate_source_sha256": json.loads((new_release / "release.json").read_text())["source_digest"]} + finally: + if hot is not None: + if hot.poll() is None: + hot.terminate() + hot.communicate(timeout=5) + if (runtime / "state.sqlite3").is_file(): + current = (self.install_root / "current").resolve() + cleanup = "from pathlib import Path;import sys;from devsquad.service import Service;s=Service(Path(sys.argv[1]));[s.cancel(r) for r in sys.argv[2:]]" + subprocess.run([str(current / "venv/bin/python"), "-P", "-c", cleanup, str(runtime), live, queued, *([council_id] if council_id else [])], + capture_output=True, env=self.environment, cwd=self.root, timeout=15) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_isolation.py b/test/core/test_council_isolation.py new file mode 100644 index 0000000..a53762d --- /dev/null +++ b/test/core/test_council_isolation.py @@ -0,0 +1,105 @@ +from __future__ import annotations + +import hashlib +import os +from pathlib import Path +import platform +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.council_isolation import command, freeze_boundary, verify_read_boundary, verify_native_bootstrap, verify_native_network +from devsquad.contracts import CapabilityUnavailable +from devsquad.mcp_server import MCPBridge + + +class CouncilIsolationTest(unittest.TestCase): + def test_bootstrap_uses_shared_cleanup_and_refuses_missing_identity_before_rpc(self): + # No provider execution: test only the exact caller authority contract. + from unittest.mock import Mock + process = Mock() + with tempfile.TemporaryDirectory(prefix="council-bootstrap-contract-") as temporary: + with patch("devsquad.diagnostics._probe_output", return_value=(0, "codex-cli test")), \ + patch("devsquad.council_isolation.command", return_value=["frozen-command"]), \ + patch("devsquad.council_isolation.subprocess.Popen", return_value=process), \ + patch("devsquad.probe_process.capture_probe_identity", return_value=None), \ + patch("devsquad.probe_process.close_probe") as close, \ + patch("devsquad.codex_protocol.JsonLinePeer") as peer: + with self.assertRaisesRegex(CapabilityUnavailable, "ownership identity"): + verify_native_bootstrap({}, executable=Path("/bin/cat"), expected_version="codex-cli test", evidence=Path(temporary)) + peer.assert_not_called() + close.assert_called_once_with(process, start_identity=None) + + def test_native_bootstrap_or_cached_catalog_is_not_network_attestation(self): + with self.assertRaisesRegex(CapabilityUnavailable, "genuine non-generating HTTPS backend response"): + verify_native_network({"probe_status": "passed", "model_list_count": 8}) + + def test_unsupported_platform_has_no_unsandboxed_fallback(self): + with patch("devsquad.council_isolation.platform.system", return_value="Linux"): + with self.assertRaises(CapabilityUnavailable): + freeze_boundary(executable=Path("/bin/cat"), evidence=ROOT) + + @unittest.skipUnless(platform.system() == "Darwin" and Path("/usr/bin/sandbox-exec").is_file(), "macOS Seatbelt required") + def test_actual_same_user_process_denies_peer_ledger_logs_artifacts_and_symlink_children(self): + with tempfile.TemporaryDirectory(prefix="council-isolation-public-") as temporary: + root = Path(temporary).resolve() + own = root / "proposer_a" + own.mkdir() + brief = own / "brief.txt" + brief.write_text("Only own frozen evidence\n") + forbidden = [] + for directory, name in (("proposer_b", "proposal.txt"), ("critic", "critique.txt"), + ("runtime", "state.sqlite3"), ("private-logs", "supervisor.log"), ("artifacts", "raw-provenance.json")): + parent = root / directory + parent.mkdir() + path = parent / name + path.write_text("Private peer/coordinator data\n") + forbidden.append(path) + boundary = freeze_boundary(executable=Path("/bin/cat"), evidence=own) + verify_read_boundary(boundary, own_file=brief, forbidden=tuple(forbidden)) + self.assertEqual(subprocess.run(command(boundary, ["/bin/cat", str(brief)]), capture_output=True).returncode, 0) + for path in forbidden: + result = subprocess.run(command(boundary, ["/bin/cat", str(path)]), capture_output=True) + self.assertNotEqual(result.returncode, 0) + self.assertNotIn(b"Private peer", result.stdout) + link = own / "peer-link" + link.symlink_to(forbidden[0]) + self.assertNotEqual(subprocess.run(command(boundary, ["/bin/cat", str(link)]), capture_output=True).returncode, 0) + # The native jail is stricter (no shell exec). Permit shell only in + # this diagnostic variant to prove the file rules inherit to children. + policy = boundary["profile"] + '(allow process-exec (literal "/bin/sh"))\n(allow file-read* (literal "/bin/sh"))\n' + child = subprocess.run(["/usr/bin/sandbox-exec", "-p", policy, "/bin/sh", "-c", 'exec /bin/cat "$1"', "probe", str(forbidden[0])], capture_output=True) + self.assertNotEqual(child.returncode, 0) + corrupt = {**boundary, "profile": boundary["profile"] + "(allow default)"} + with self.assertRaises(CapabilityUnavailable): + command(corrupt, ["/bin/cat", str(brief)]) + # The native dependency additions must preserve these exact denials. + native = freeze_boundary(executable=Path("/bin/cat"), evidence=own, native_codex=True) + verify_read_boundary(native, own_file=brief, forbidden=tuple(forbidden)) + self.assertNotEqual(subprocess.run(command(native, ["/bin/cat", str(link)]), capture_output=True).returncode, 0) + environment = {"PATH": "/usr/bin:/bin"} # no Council/worker marker + self.assertNotEqual(subprocess.run(command(native, ["/bin/cat", str(forbidden[2])]), + capture_output=True, env=environment).returncode, 0) + # Alternate MCP servers cannot escape by spawning a new interpreter: + # it is not a frozen execution dependency, even under the same uid. + alternate = subprocess.run(["/usr/bin/sandbox-exec", "-p", native["profile"], sys.executable, + "-c", "import pathlib; pathlib.Path(__import__('sys').argv[1]).read_bytes()", str(forbidden[2])], + capture_output=True, env=environment) + self.assertNotEqual(alternate.returncode, 0) + + def test_worker_mcp_saved_reads_cannot_bypass_role_directory(self): + with tempfile.TemporaryDirectory(prefix="council-mcp-denial-") as directory: + bridge = MCPBridge(Path(directory)) + with patch.dict(os.environ, {"DEVSQUAD_WORKER": "1", "DEVSQUAD_DELEGATION_DEPTH": "1", "DEVSQUAD_COUNCIL_ROLE": "proposer_a"}): + for operation in (lambda: bridge.status("peer-run"), lambda: bridge.events("peer-run"), lambda: bridge.result("peer-run")): + result = operation() + self.assertFalse(result["ok"]) + self.assertEqual(result["error"]["code"], "POLICY_DENIED") + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_reconciliation.py b/test/core/test_council_reconciliation.py new file mode 100644 index 0000000..1750423 --- /dev/null +++ b/test/core/test_council_reconciliation.py @@ -0,0 +1,229 @@ +"""Public fixture/process Council gates against the shared R6 authority.""" +from __future__ import annotations + +import contextlib +from datetime import datetime, timedelta, timezone +import io +import json +import os +from pathlib import Path +import shlex +import sys +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path[:0] = [str(ROOT / "plugin/core/src"), str(ROOT / "test/core")] +import test_council_runtime as fixtures +from devsquad import cli, detached +from devsquad.contracts import ContractError +from devsquad.service import Service +from devsquad.store import ConflictError, Store, request_hash +from devsquad.supervisor import _live_group_exists + + +class CouncilReconciliationTest(unittest.TestCase): + setUp = fixtures.CouncilRuntimeTest.setUp + git = fixtures.CouncilRuntimeTest.git + wait = fixtures.CouncilRuntimeTest.wait + receipt = fixtures.CouncilRuntimeTest.receipt + + def host_run(self, key): + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, key, _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + return started["run_id"] + + def invoke(self, argv): + output = io.StringIO() + with contextlib.redirect_stdout(output): + code = cli.main([*argv, "--runtime-dir", str(self.runtime)]) + return code, output.getvalue() + + def finish(self, run): + return self.service.finish(run, "accept", "--frozen 'choice'", chosen="synthesis", + supported_claims=["--no blind retry"], discarded_alternatives=["'blind' retry"], validation="--duplicate test") + + def test_human_omitted_id_explicit_choice_and_exact_retry_round_trip(self): + run = self.host_run("normal-council-finish") + code, output = self.invoke(["status", "--project-dir", str(self.repo)]) + self.assertEqual(code, 0, output) + self.assertIn("Proposal A:", output) + self.assertIn("Proposal B:", output) + self.assertIn("Dissent retention", output) + self.assertIn("explicit disposition, --choose", output) + self.assertNotIn("--accept --reason=", output) # no automatic choice + code, output = self.invoke(["finish", "--project-dir", str(self.repo), "--accept", "--reason", "no choice"]) + self.assertEqual(code, 64, output) + before = self.service.status(run)["version"] + with patch.object(self.service, "handoff_complete", side_effect=RuntimeError("claim committed")): + with self.assertRaises(RuntimeError): + self.finish(run) + version = self.service.status(run)["version"] + self.assertEqual(version, before + 1) + code, output = self.invoke(["status", run]) + self.assertEqual(code, 0, output) + command = shlex.split(next(line[6:] for line in output.splitlines() if line.startswith("Next: "))) + self.assertIn("--choose=synthesis", command) + self.assertIn("--supported-claim=--no blind retry", command) + code, output = self.invoke(["status", run, "--json"]) + self.assertNotIn("handoff_view", json.loads(output)["data"]) + code, output = self.invoke(command[1:]) + self.assertEqual(code, 0, output) + self.assertEqual(self.receipt(run)["lead"]["choice"]["validation"], "--duplicate test") + code, output = self.invoke(["result", "--project-dir", str(self.repo)]) + self.assertEqual(code, 0, output) + self.assertIn("receipt.json:", output) + + def test_host_expiry_rejection_is_audited_then_recovers_exact_intent(self): + run = self.host_run("host-expiry-at-submission") + captured = {} + original_complete = self.service.handoff_complete + def expire(run_id, claim, decision): + captured.update(claim=claim, decision=decision) + captured["later"] = datetime.fromisoformat(claim["expires_at"]) + timedelta(seconds=1) + with patch("devsquad.store._authoritative_now", return_value=captured["later"]): + return original_complete(run_id, claim, decision) + with patch.object(self.service, "handoff_complete", side_effect=expire): + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.finish(run) + store = self.service._store() + try: + rejected = dict(store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (captured["claim"]["handoff_id"],)).fetchone()) + marker = store.terminal_finish_decision(run, captured["claim"]["handoff_id"]) + self.assertEqual(marker, captured["decision"]) + finally: + store.close() + with patch("devsquad.store._authoritative_now", return_value=captured["later"]): + with self.assertRaisesRegex(ConflictError, "different terminal finish intent"): + self.service.finish(run, "reject", "changed intent", chosen="A", validation="check") + self.assertEqual(Service(self.runtime).resume(run)["state"], "succeeded") + store = self.service._store() + try: + events = store.events_for_run(run) + recovered = [e for e in events if e["type"] == "handoff.completion_recovered"] + self.assertEqual(len(recovered), 1) + self.assertEqual(recovered[0]["payload"]["rejected_submission"], rejected) + self.assertEqual(recovered[0]["payload"]["rejected_submission_sha256"], request_hash(rejected)) + self.assertEqual(recovered[0]["payload"]["fencing_token"], captured["claim"]["fencing_token"] + 1) + self.assertEqual(len(store.outcomes_for_run(run)), 1) + finally: + store.close() + + def test_guided_finish_never_adopts_app_claim_even_same_owner_expired(self): + run = self.host_run("app-claim-no-terminal-authority") + claim = self.service.handoff_claim(run, self.service.status(run)["version"], "terminal-operator")["claim"] + for now in (datetime.now(timezone.utc), datetime.fromisoformat(claim["expires_at"]) + timedelta(seconds=1)): + version = self.service.status(run)["version"] + with patch("devsquad.store._authoritative_now", return_value=now): + with self.assertRaisesRegex(ConflictError, "already has a host claim"): + self.finish(run) + with self.assertRaisesRegex(ConflictError, "no exact current"): + Service(self.runtime).resume(run) + self.assertEqual(self.service.status(run)["version"], version) + self.service.cancel(run) + + def controlled_stage(self, run_id): + # Scheduling seam only: public start + real durable subprocess workers; + # coordinator auto-resume is suppressed at the exact durable barrier. + store = self.service._store() + try: + run = store.run(run_id) + finally: + store.close() + with patch.dict(os.environ, {"PYTHONPATH": run["package_path"]}), patch.object(Service, "resume", return_value={}): + self.assertEqual(detached.main(["--database", str(self.service.database), "--artifacts", str(self.service.artifacts), + "--run-id", run_id, "--expected-version", str(run["version"]), "--package-digest", run["package_digest"]]), 0) + + def headless_at_imported_lead(self, key): + self.task["lead"]["mode"] = "headless" + with patch.object(Service, "_spawn_daemon", return_value=0): + started = self.service.start(self.task, key, _internal_council_fixture=self.fixture) + for _ in range(3): + self.controlled_stage(started["run_id"]) + self.assertEqual(self.service.resume(started["run_id"])["state"], "queued") + self.controlled_stage(started["run_id"]) + return started["run_id"] + + def test_headless_claim_and_submitted_crashes_recover_without_extra_workers(self): + for boundary in ("record_handoff_submission", "complete_handoff_terminal"): + for expired in (False, True): + with self.subTest(boundary=boundary, expired=expired): + run = self.headless_at_imported_lead(f"headless-{boundary}-{expired}") + with patch.object(Store, boundary, side_effect=RuntimeError("after durable boundary")): + with self.assertRaises(RuntimeError): + self.service.resume(run) + store = self.service._store() + try: + handoff = store.handoff_snapshot(run) + old = dict(store.connection.execute("SELECT * FROM claims WHERE run_id=?", (run,)).fetchone()) + submitted = store.recorded_handoff_submission(run, handoff.handoff_id) + finally: + store.close() + future = datetime.fromisoformat(old["lease_expires_at"]) + timedelta(seconds=1) + class FutureDateTime(datetime): + @classmethod + def now(cls, tz=None): + return future + with contextlib.ExitStack() as stack: + if expired: + stack.enter_context(patch("devsquad.council_runtime.datetime", FutureDateTime)) + stack.enter_context(patch("devsquad.store._authoritative_now", return_value=future)) + result = Service(self.runtime).resume(run) + self.assertEqual(result["state"], "succeeded") + receipt = self.receipt(run) + self.assertEqual(receipt["worker_invocations"], 4) + self.assertEqual(receipt["identity_scope"], "all_fixture") + store = self.service._store() + try: + attempts = store.attempts_for_run(run) + self.assertTrue(all(not _live_group_exists(a["pgid"]) for a in attempts)) + latest = dict(store.connection.execute("SELECT * FROM claims WHERE run_id=?", (run,)).fetchone()) + # Recorded submissions need no new lease or fence. + self.assertEqual(latest["fencing_token"], old["fencing_token"] + int(expired and submitted is None)) + self.assertFalse(latest["active"]) + self.assertEqual(len(store.outcomes_for_run(run)), 1) + finally: + store.close() + + def test_headless_expired_submission_is_retained_and_new_fence_completes_saved_lead(self): + run = self.headless_at_imported_lead("headless-expiry-at-submission") + with self.assertRaisesRegex(ConflictError, "does not accept a host claim"): + self.service.handoff_claim(run, self.service.status(run)["version"], "app-owner") + captured = {} + original_record = Store.record_handoff_submission + def expire(store, run_id, claim, decision, **kwargs): + captured.update(claim=claim, decision=decision) + captured["later"] = datetime.fromisoformat(claim.expires_at) + timedelta(seconds=1) + return original_record(store, run_id, claim, decision, now=captured["later"]) + with patch.object(Store, "record_handoff_submission", new=expire): + with self.assertRaisesRegex(ConflictError, "expired_claim"): + self.service.resume(run) + future = captured["later"] + class FutureDateTime(datetime): + @classmethod + def now(cls, tz=None): + return future + with patch("devsquad.council_runtime.datetime", FutureDateTime), patch("devsquad.store._authoritative_now", return_value=future): + self.assertEqual(Service(self.runtime).resume(run)["state"], "succeeded") + store = self.service._store() + try: + submissions = [dict(row) for row in store.connection.execute("SELECT * FROM handoff_submissions WHERE handoff_id=?", (captured["claim"].handoff_id,))] + self.assertEqual(len(submissions), 2) + rejected = next(row for row in submissions if row["outcome"] == "rejected") + recorded = next(row for row in submissions if row["outcome"] == "recorded") + self.assertEqual(rejected["rejection_code"], "expired_claim") + self.assertEqual(recorded["fencing_token"], rejected["fencing_token"] + 1) + self.assertNotEqual(recorded["submission_id"], rejected["submission_id"]) + self.assertEqual(json.loads(recorded["decision_json"])["council_choice"], captured["decision"]["council_choice"]) + self.assertEqual(len(store.attempts_for_run(run)), 4) + self.assertEqual(len(store.outcomes_for_run(run)), 1) + self.assertTrue(all(not _live_group_exists(a["pgid"]) for a in store.attempts_for_run(run))) + with self.assertRaisesRegex(ConflictError, "expired_claim"): + store.record_handoff_submission(run, captured["claim"], captured["decision"]) + finally: + store.close() + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_council_runtime.py b/test/core/test_council_runtime.py new file mode 100644 index 0000000..14254d6 --- /dev/null +++ b/test/core/test_council_runtime.py @@ -0,0 +1,314 @@ +from __future__ import annotations + +import copy +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import time +import unittest +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "plugin/core/src")) +from devsquad.service import Service +from devsquad.store import Store, request_hash, ConflictError +from devsquad.contracts import ContractError +from devsquad.council_worker import role_packet +from devsquad.contracts import ExecutionIdentity, LaunchSpec +from devsquad.supervisor import Supervisor, _live_group_exists + + +class CouncilRuntimeTest(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="devsquad-council-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.repo = self.root / "repo" + self.runtime = self.root / "runtime" + self.repo.mkdir() + self.git("init", "-q") + self.git("config", "user.email", "fixture@example.invalid") + self.git("config", "user.name", "Public controlled fixture") + (self.repo / "README").write_text("A retry duplicates a non-idempotent request.\n") + profiles = [] + for role, model in (("proposer_a", "model-a"), ("proposer_b", "model-b"), ("critic", "model-c")): + profiles.append({"id": role, "harness": "fixture", "model_family": model, + "model_id": model, "effort": {"value": "low", "transport": "native"}, + "required_tools": [], "permission_policy": "read_only", "account_pool_id": "shared", + "billing_mode": "subscription", "quality_status": "proven", "evidence_refs": ["public-fixture"]}) + (self.repo / "profiles.json").write_text(json.dumps({"schema_version": 1, "profiles": profiles, "bindings": {}})) + policy = {"schema_version": 1, "id": "controlled-council", "version": 1, + "roles": {role: [{"kind": "profile", "id": role}] for role in ("proposer_a", "proposer_b", "critic")}, + "task_classes": {"fixture-council": "proven"}, "require_different_model_for_review": True, + "prefer_different_harness_for_review": True, "account_pools": {"shared": { + "allowed_billing_modes": ["subscription"], "max_concurrency": 1, + "unknown_capacity_policy": "allow_bounded"}}, "experiment_budget": {}} + policy["roles"]["lead"] = [{"kind": "profile", "id": "critic"}] + (self.repo / "policy.json").write_text(json.dumps(policy)) + self.git("add", ".") + self.git("commit", "-qm", "public frozen input") + self.source_oid = self.git("rev-parse", "HEAD") + self.task = {"schema_version": 1, "project": {"repo_path": str(self.repo), "base_ref": "HEAD", "target_ref": "HEAD"}, + "workflow": "council-decision", "goal": "Choose a safe retry policy", "task_class": "fixture-council", + "acceptance": [{"id": "safe-retry", "description": "Avoid duplicate side effects", "evidence_kind": "review"}], + "checks": [{"id": "public-check", "argv": ["git", "diff", "--check"], "cwd": ".", "timeout_seconds": 5, "required_to_pass": True}], + "scope": {"read_paths": ["README"], "write_paths": []}, "lead": {"mode": "headless"}, + "routing": {"profiles_file": "profiles.json", "policy_file": "policy.json"}, + "budget": {"wall_seconds": 60, "max_worker_invocations": 4, "max_revisions": 0, "max_fallbacks_per_step": 0}, + "origin": {"surface": "cli"}, "council": {"schema_version": 1, "enabled": True, "automatic": False, + "reason": "Compare independent retry alternatives", "min_valid_proposals": 2, "required_critics": 1, + "max_invocations": 4, "seed": "a" * 64, "evidence": [], + "rubric": [{"id": "safety", "description": "Does not duplicate side effects"}]}} + proposal = {"summary": "Use an idempotency key", "approach": "Bounded retries with a stable key", + "claims": [{"text": "A stable key avoids duplicate effects", "evidence_ids": []}], + "validation": "Exercise a repeated request"} + self.fixture = {"proposer_a": {"document": proposal}, "proposer_b": {"document": dict(proposal, summary="Do not retry blindly")}, + "critic": {"document": {"summary": "Both avoid a blind retry", "assessments": [ + {"label": label, "criterion_id": "safety", "status": "supported", "reason": "Requires duplicate protection", "evidence_ids": []} for label in ("A", "B")], + "objections": [{"id": "retention", "label": "A", "reason": "Key retention must cover retries", "evidence_ids": []}]}}, + "lead": {"document": {"disposition": "accept", "chosen": "synthesis", "reason": "Retain both safeguards", + "supported_claims": ["Never retry a non-idempotent request blindly"], "discarded_alternatives": ["Blind retry"], + "unresolved_objections": ["retention"], "validation": "Test duplicates and retention expiry"}}} + self.service = Service(self.runtime) + + def git(self, *args): + return subprocess.run(["git", "-C", str(self.repo), *args], check=True, capture_output=True, text=True).stdout.strip() + + def wait(self, run_id, states, timeout=12): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = self.service.status(run_id) + if status["state"] in states: + return status + if status["state"] == "awaiting_host" and self.task["lead"]["mode"] == "headless": + try: + self.service.resume(run_id) + except ConflictError: + pass # Another detached coordinator may win this exact fence. + time.sleep(.05) + self.fail(f"Council did not reach {states}: {self.service.status(run_id)}") + + def receipt(self, run_id): + result = self.service.result(run_id) + artifact = next(a for a in result["artifacts"] if a["name"] == "receipt.json") + return json.loads(Path(artifact["path"]).read_bytes()) + + def test_public_process_flow_seals_proposals_and_preserves_dissent_and_unknown_usage(self): + started = self.service.start(self.task, "public-council", _internal_council_fixture=self.fixture) + self.assertEqual(self.wait(started["run_id"], {"succeeded", "failed"})["state"], "succeeded") + receipt = self.receipt(started["run_id"]) + self.assertEqual([a["role"] for a in receipt["attempts"]], ["proposer_a", "proposer_b", "critic", "lead"]) + self.assertEqual(receipt["worker_invocations"], 4) + self.assertEqual(receipt["identity_scope"], "all_fixture") + self.assertTrue(all(a["usage"] is None and a["observed_identity"] is None for a in receipt["attempts"])) + self.assertEqual(receipt["dissent"][0]["id"], "retention") + self.assertFalse(receipt["automatic_enabled"]) + self.assertEqual(self.git("rev-parse", "HEAD"), self.source_oid) + self.assertEqual(self.git("status", "--porcelain"), "") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + self.addCleanup(store.close) + snapshot = json.loads(store.run(started["run_id"])["mutable_snapshot"]) + outcomes = store.outcomes_for_run(started["run_id"]) + self.assertEqual(len(outcomes), 1) + outcome = outcomes[0]["outcome"] + self.assertEqual(outcome["kind"], "final") + self.assertEqual({item["role"] for item in outcome["contributions"]}, {"proposer_a", "proposer_b", "critic", "lead"}) + self.assertTrue(all(item["status"] == "passed" for item in outcome["criteria"])) + self.service.result(started["run_id"]) + self.service.learning_report(str(self.repo)) + self.assertEqual(len(store.outcomes_for_run(started["run_id"])), 1) + for role in ("proposer_a", "proposer_b"): + packet = role_packet(snapshot, role) + self.assertEqual(set(packet), {"brief"}) + self.assertNotIn("profile", json.dumps(packet)) + critic = role_packet(snapshot, "critic") + self.assertEqual(set(critic["proposals"]), {"A", "B"}) + self.assertNotIn("model-a", json.dumps(critic)) + self.assertNotIn("model-b", json.dumps(critic)) + + def test_invalid_critic_cannot_become_consensus(self): + fixture = copy.deepcopy(self.fixture) + fixture["critic"]["document"]["assessments"].pop() + started = self.service.start(self.task, "invalid-critic", _internal_council_fixture=fixture) + self.assertEqual(self.wait(started["run_id"], {"failed"})["state"], "failed") + receipt = self.receipt(started["run_id"]) + self.assertEqual(receipt["worker_invocations"], 3) + self.assertIsNone(receipt["lead"]["disposition"]) + + def test_cancel_owned_gated_launcher_never_publishes_terminal_before_cleanup(self): + with patch.object(self.service, "_spawn_daemon", return_value=0): + started = self.service.start(self.task, "gated-owned-cancel", _internal_council_fixture=self.fixture) + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + self.addCleanup(store.close) + run = store.run(started["run_id"]) + snapshot = json.loads(run["mutable_snapshot"]) + selected = snapshot["routing"]["roles"]["proposer_a"]["selected"] + identity = ExecutionIdentity("fixture", "controlled", None, selected["profile"]["model_family"], + selected["profile"]["model_id"], "low", permissions="read_only", account_pool="shared") + spec = LaunchSpec(1, "fixture", "cli_exec", (sys.executable, "-P", "-m", "devsquad.fake_step", "--delay", "30"), + run["worktree_path"], None, 5, identity, + {"DEVSQUAD_WORKER": "1", "DEVSQUAD_COUNCIL_ROLE": "proposer_a"}) + original_release = Supervisor._release_runner_gate + owned = [] + + def cancel_before_gate_release(descriptor): + attempt = store.attempt(started["run_id"]) + owned.append(attempt["pgid"]) + self.assertTrue(_live_group_exists(attempt["pgid"])) + cancelled = self.service.cancel(started["run_id"]) + self.assertEqual(cancelled["state"], "cancelling") + self.assertIsNone(store.artifact_named(started["run_id"], "receipt.json")) + original_release(descriptor) + + supervisor = Supervisor(store) + with patch.dict(os.environ, {"PYTHONPATH": str(Path(run["package_path"]))}), patch.object(Supervisor, "_release_runner_gate", side_effect=cancel_before_gate_release): + handle = supervisor.launch_durable(started["run_id"], run["version"], spec, "controlled-gated-owner", run["package_digest"], + role="proposer_a", profile_id=selected["profile_id"], profile_index=0) + supervisor.wait_durable(handle, 5) + self.assertEqual(self.service.status(started["run_id"])["state"], "cancelled") + self.assertTrue(all(not _live_group_exists(group) for group in owned)) + self.assertIsNone(self.receipt(started["run_id"])["lead"]["disposition"]) + + def test_failed_mandatory_check_cannot_be_overridden_by_lead(self): + self.task["checks"][0]["argv"] = ["false"] + started = self.service.start(self.task, "failed-check", _internal_council_fixture=self.fixture) + self.assertEqual(self.wait(started["run_id"], {"failed"})["state"], "failed") + receipt = self.receipt(started["run_id"]) + self.assertIsNone(receipt["lead"]["disposition"]) + + def test_host_exact_claim_choice_and_terminal_replay(self): + self.task["lead"]["mode"] = "host" + self.task["budget"]["max_worker_invocations"] = 3 + started = self.service.start(self.task, "host", _internal_council_fixture=self.fixture) + waiting = self.wait(started["run_id"], {"awaiting_host"}) + acquired = self.service.handoff_claim(started["run_id"], waiting["version"], "fixture-host") + packet = acquired["handoff"]["packet"] + choice = self.fixture["lead"]["document"] + body = {"schema_version": 1, "submission_id": "host-choice", "disposition": "accept", "reason": choice["reason"], + "evidence_refs": [{"artifact_id": a["artifact_id"], "sha256": a["sha256"]} for a in packet["artifacts"]], "council_choice": choice} + decision = {**body, "submission_hash": request_hash(body)} + result = self.service.handoff_complete(started["run_id"], acquired["claim"], decision) + self.assertEqual(result["state"], "succeeded") + self.assertTrue(self.service.handoff_complete(started["run_id"], acquired["claim"], decision)["replayed"]) + + def test_guided_host_finish_has_explicit_choice_and_no_json(self): + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, "guided-host", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + view = self.service.council_handoff_view(started["run_id"]) + self.assertIn("Council handoff", view["report"]) + result = self.service.finish_council(started["run_id"], "accept", "retain safeguards", chosen="synthesis", + supported_claims=["No blind retries"], discarded_alternatives=["Blind retry"], validation="Test retention expiry") + self.assertEqual(result["state"], "succeeded") + with self.assertRaises(ConflictError): + self.service.finish_council(started["run_id"], "accept", "replay", chosen="A", + supported_claims=["claim"], discarded_alternatives=[], validation="check") + + def test_missing_proposer_empty_output_and_missing_critic_never_form_quorum(self): + for index, (role, empty) in enumerate((("proposer_a", False), ("proposer_b", True), ("critic", False))): + fixture = copy.deepcopy(self.fixture) + if empty: + fixture[role] = {"document": {}} + else: + fixture.pop(role) + started = self.service.start(self.task, f"missing-{index}", _internal_council_fixture=fixture) + self.assertEqual(self.wait(started["run_id"], {"failed"})["state"], "failed") + receipt = self.receipt(started["run_id"]) + self.assertIsNone(receipt["lead"]["disposition"]) + self.assertLess(receipt["worker_invocations"], 4) + + def test_cancel_active_role_and_saved_host_wait_never_accept(self): + fixture = copy.deepcopy(self.fixture) + fixture["proposer_a"]["delay_seconds"] = 4 + started = self.service.start(self.task, "active-cancel", _internal_council_fixture=fixture) + self.wait(started["run_id"], {"running"}) + self.service.cancel(started["run_id"]) + self.assertEqual(self.wait(started["run_id"], {"cancelled"})["state"], "cancelled") + self.assertIsNone(self.receipt(started["run_id"])["lead"]["disposition"]) + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, "waiting-cancel", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + self.assertEqual(self.service.cancel(started["run_id"])["state"], "cancelled") + self.assertEqual(self.receipt(started["run_id"])["worker_invocations"], 3) + + def test_queued_cancel_and_restart_use_saved_exact_snapshot(self): + with patch.object(self.service, "_spawn_daemon"): + started = self.service.start(self.task, "queued-cancel", _internal_council_fixture=self.fixture) + self.assertEqual(self.service.cancel(started["run_id"])["state"], "cancelled") + self.assertEqual(self.receipt(started["run_id"])["worker_invocations"], 0) + with patch.object(self.service, "_spawn_daemon"): + started = self.service.start(self.task, "restart", _internal_council_fixture=self.fixture) + self.service = Service(self.runtime) + self.service.resume(started["run_id"]) + self.assertEqual(self.wait(started["run_id"], {"succeeded", "failed"})["state"], "succeeded") + + def test_guided_claim_crash_and_submitted_crash_resume_exact_decision(self): + self.task["lead"]["mode"] = "host" + for boundary in ("record_handoff_submission", "complete_handoff_terminal"): + started = self.service.start(self.task, f"crash-{boundary}", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + with patch.object(Store, boundary, side_effect=RuntimeError("injected crash")): + with self.assertRaises(RuntimeError): + self.service.finish_council(started["run_id"], "accept", "frozen crash choice", chosen="A", + supported_claims=["No blind retry"], discarded_alternatives=["Blind retry"], validation="Duplicate request test") + self.service = Service(self.runtime) + result = self.service.resume(started["run_id"]) + self.assertEqual(result["state"], "succeeded") + self.assertEqual(self.receipt(started["run_id"])["lead"]["choice"]["reason"], "frozen crash choice") + + def test_expired_guided_claim_reacquires_only_its_frozen_intent(self): + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, "expired-guided", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + with patch.object(Store, "record_handoff_submission", side_effect=RuntimeError("crash after claim")): + with self.assertRaises(RuntimeError): + self.service.finish_council(started["run_id"], "accept", "intent before expiry", chosen="B", + supported_claims=["claim"], discarded_alternatives=[], validation="check") + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + from datetime import datetime, timedelta, timezone + expires = store.handoff_snapshot(started["run_id"]).claim.expires_at + store.close() + future = datetime.fromisoformat(expires) + timedelta(seconds=1) + class FutureDateTime(datetime): + @classmethod + def now(cls, tz=None): + return future + with patch("devsquad.council_runtime.datetime", FutureDateTime), patch("devsquad.store._authoritative_now", return_value=future): + self.assertEqual(Service(self.runtime).resume(started["run_id"])["state"], "succeeded") + + def test_same_owner_low_level_takeover_does_not_recover_old_guided_intent(self): + self.task["lead"]["mode"] = "host" + started = self.service.start(self.task, "intent-race", _internal_council_fixture=self.fixture) + self.wait(started["run_id"], {"awaiting_host"}) + with patch.object(Store, "record_handoff_submission", side_effect=RuntimeError("crash")): + with self.assertRaises(RuntimeError): + self.service.finish_council(started["run_id"], "accept", "old intent", chosen="A", + supported_claims=["claim"], discarded_alternatives=[], validation="check") + from datetime import datetime, timedelta + store = Store(self.runtime / "state.sqlite3", self.runtime / "artifacts") + future = datetime.fromisoformat(store.handoff_snapshot(started["run_id"]).claim.expires_at) + timedelta(seconds=1) + store.close() + with patch("devsquad.store._authoritative_now", return_value=future): + version = self.service.status(started["run_id"])["version"] + self.service.handoff_claim(started["run_id"], version, "terminal-operator") + with self.assertRaises(ConflictError): + self.service.resume(started["run_id"]) + self.assertEqual(self.service.status(started["run_id"])["state"], "awaiting_host") + + def test_preflight_quota_exhaustion_is_zero_launches_not_consensus(self): + from datetime import datetime, timedelta, timezone + now = datetime.now(timezone.utc) + self.service.capacity_observe({"schema_version": 1, "observation_id": "quota-zero", "pool_id": "shared", "window_id": "subscription", + "applies_to": {"harnesses": [], "model_families": [], "model_ids": []}, + "source": "native_reported", "observed_at": now.isoformat(), "expires_at": (now + timedelta(minutes=5)).isoformat(), + "used": 100, "limit": 100, "unit": "percent", "resets_at": (now + timedelta(minutes=5)).isoformat(), "confidence": "confirmed"}) + started = self.service.start(self.task, "quota", _internal_council_fixture=self.fixture) + self.assertEqual(started["state"], "failed") + self.assertEqual(self.receipt(started["run_id"])["worker_invocations"], 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/core/test_decision_store.py b/test/core/test_decision_store.py index b40704b..dcea3ca 100644 --- a/test/core/test_decision_store.py +++ b/test/core/test_decision_store.py @@ -191,7 +191,8 @@ def test_schema_thirteen_contains_decision_cache_and_run_links(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 16) + from devsquad.store import SUPPORTED_SCHEMA_VERSION + self.assertEqual(version, SUPPORTED_SCHEMA_VERSION) tables = { row[0] for row in self.store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_diagnostics.py b/test/core/test_diagnostics.py index 3121ef8..9c01af3 100644 --- a/test/core/test_diagnostics.py +++ b/test/core/test_diagnostics.py @@ -4,6 +4,7 @@ import os from pathlib import Path import signal +import shutil import sys import tempfile import time @@ -41,7 +42,8 @@ def setUp(self): self.core_patch.start() self.addCleanup(self.core_patch.stop) - def install(self, name, *, version=None, response=None, returncode=0, raw=None, delay=0): + def install(self, name, *, version=None, response=None, returncode=0, raw=None, delay=0, + version_exit_delay=0, auth_exit_delay=0): versions = { "codex": "codex-cli 0.159.2", "claude": "2.1.220 (Claude Code)", "grok": "1.0.46", "antigravity": "1.2.14", @@ -51,6 +53,7 @@ def install(self, name, *, version=None, response=None, returncode=0, raw=None, configuration = { "name": name, "version": version, "response": response, "returncode": returncode, "raw": raw, "log": str(self.log), "delay": delay, + "version_exit_delay": version_exit_delay, "auth_exit_delay": auth_exit_delay, } binary.write_text("#!" + sys.executable + "\n" + """ import json, os, sys, time @@ -60,11 +63,19 @@ def record(value): output.write(json.dumps(value) + '\\n') record({'argv': sys.argv[1:], 'environment': dict(os.environ)}) if sys.argv[1:] == ['--version']: - print(configuration['version']) + print(configuration['version'], flush=True) + if configuration['version_exit_delay']: + os.close(1) + time.sleep(configuration['version_exit_delay']) + os._exit(0) elif sys.argv[1:] == ['auth', 'status', '--json']: time.sleep(configuration['delay']) - print(configuration['raw'] if configuration['raw'] is not None else json.dumps(configuration['response'])) + print(configuration['raw'] if configuration['raw'] is not None else json.dumps(configuration['response']), flush=True) print('private-stderr-token', file=sys.stderr) + if configuration['auth_exit_delay']: + os.close(1) + time.sleep(configuration['auth_exit_delay']) + os._exit(configuration['returncode']) sys.exit(configuration['returncode']) elif sys.argv[1:] == ['app-server', '--listen', 'stdio://']: for line in sys.stdin: @@ -163,6 +174,9 @@ def test_authentication_and_registration_do_not_invent_operation_proof(self): self.assertTrue(report["ready"]) self.assertTrue(report["supported_workflows"]["issue-delivery"]["ready"]) self.assertFalse(report["supported_workflows"]["council"]["ready"]) + self.assertTrue(report["supported_workflows"]["council"]["implemented_partial"]) + self.assertFalse(report["supported_workflows"]["council"]["native_ready"]) + self.assertEqual(report["supported_workflows"]["council"]["reason"], "native_network_attestation_unavailable") for row in report["adapters"]: self.assertIs(row["authenticated"], True) self.assertIsNone(row["operation_verified"]) @@ -334,7 +348,7 @@ def capture_process(*args, **kwargs): self.assertRaises(TimeoutError), ): diagnostics._probe_output( - [sys.executable, "-c", code], project=self.project, + [sys.executable, "-B", "-c", code], project=self.project, environment=diagnostics._environment(self.home), ) self.assertLess(time.monotonic() - started, 3) @@ -402,6 +416,113 @@ def test_known_capture_missing_observation_has_no_unproven_group_authority(self) process.wait.assert_not_called() process.stdout.close.assert_called_once() + def test_version_eof_preserves_delayed_natural_success_and_retained_anchor(self): + binary = self.install("codex", version_exit_delay=0.15) + processes = [] + real_spawn, real_close = diagnostics.subprocess.Popen, diagnostics._close_probe + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + if process.args == [str(binary), "--version"]: + processes.append(process) + return process + def close(process, **kwargs): + if process in processes: + self.assertIsNone(process.returncode, "natural-exit observation reaped the ownership anchor") + real_close(process, **kwargs) + started = time.monotonic() + with ( + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + mock.patch.object(diagnostics, "_close_probe", side_effect=close), + ): + code, output = diagnostics._probe_output([str(binary), "--version"], + project=self.project, environment=diagnostics._environment(self.home)) + self.assertEqual((code, output.strip()), (0, "codex-cli 0.159.2")) + self.assertGreaterEqual(time.monotonic() - started, 0.15) + self.assertEqual(processes[0].returncode, 0) + self.assertFalse(_live_group_exists(processes[0].pid)) + self.assertTrue(processes[0].stdout.closed) + + def test_claude_auth_eof_preserves_delayed_natural_logged_in_and_logged_out_codes(self): + for logged_in, returncode in ((True, 0), (False, 1)): + with self.subTest(logged_in=logged_in): + method = "claude.ai" if logged_in else "none" + binary = self.install("claude", response={"loggedIn": logged_in, "authMethod": method}, + returncode=returncode, auth_exit_delay=0.15) + processes = [] + real_spawn, real_close = diagnostics.subprocess.Popen, diagnostics._close_probe + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + if process.args == [str(binary), "auth", "status", "--json"]: + processes.append(process) + return process + def close(process, **kwargs): + if process in processes: + self.assertIsNone(process.returncode, "natural-exit observation reaped the ownership anchor") + real_close(process, **kwargs) + with ( + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + mock.patch.object(diagnostics, "_close_probe", side_effect=close), + ): + report = self.report() + row = self.row(report, "claude") + self.assertTrue(row["supported"]) + self.assertIs(row["authenticated"], logged_in) + self.assertEqual(row["ready"], logged_in) + self.assertEqual(processes[0].returncode, returncode) + self.assertFalse(_live_group_exists(processes[0].pid)) + self.assertTrue(processes[0].stdout.closed) + + def test_hung_after_auth_eof_times_out_without_false_success_and_cleans_child(self): + binary = self.install("claude", response={"loggedIn": True, "authMethod": "claude.ai"}, + auth_exit_delay=60) + processes = [] + real_spawn = diagnostics.subprocess.Popen + def spawn(*args, **kwargs): + process = real_spawn(*args, **kwargs) + if process.args == [str(binary), "auth", "status", "--json"]: + processes.append(process) + return process + started = time.monotonic() + with ( + mock.patch.object(diagnostics, "PROBE_TIMEOUT_SECONDS", 0.2), + mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), + ): + authentication = diagnostics._claude_auth(str(binary), project=self.project, + environment=diagnostics._environment(self.home)) + self.assertIsNone(authentication["authenticated"]) + self.assertFalse(authentication["subscription_supported"]) + self.assertGreaterEqual(time.monotonic() - started, 0.2) + self.assertLess(time.monotonic() - started, 2.5) + self.assertIsNotNone(processes[0].returncode) + self.assertFalse(_live_group_exists(processes[0].pid)) + self.assertTrue(processes[0].stdout.closed) + + def test_controlled_python_import_does_not_write_payload_bytecode_with_minimal_env(self): + source = self.root / "private-source" + shutil.copytree(ROOT / "plugin/core/src/devsquad", source / "devsquad", + ignore=shutil.ignore_patterns("__pycache__", "*.pyc")) + before = {str(path.relative_to(source)): path.read_bytes() + for path in source.rglob("*") if path.is_file()} + code = ( + "import os,sys\n" + "assert sys.dont_write_bytecode\n" + "assert 'PYTHONDONTWRITEBYTECODE' not in os.environ\n" + f"sys.path.insert(0,{str(source)!r})\n" + "from devsquad.supervisor import process_start_identity\n" + "assert process_start_identity(os.getpid()) is not None\n" + "print('codex-cli 0.159.2',flush=True)\n" + ) + environment = diagnostics._environment(self.home) + self.assertEqual(set(environment), {"HOME", "USER", "PATH"}) + result = diagnostics._probe_output([sys.executable, "-B", "-c", code], + project=self.project, environment=environment) + self.assertEqual(result, (0, "codex-cli 0.159.2\n")) + after = {str(path.relative_to(source)): path.read_bytes() + for path in source.rglob("*") if path.is_file()} + self.assertEqual(before, after) + self.assertFalse(list(source.rglob("__pycache__"))) + self.assertFalse(list(source.rglob("*.pyc"))) + def test_probe_does_not_read_output_without_captured_identity(self): process = mock.Mock(pid=987654, returncode=None, stderr=None) process.wait.return_value = 0 @@ -448,7 +569,7 @@ def test_unidentified_live_child_is_reaped_without_group_signals(self): "time.sleep(60)\n" ) process = diagnostics.subprocess.Popen( - [sys.executable, "-c", code], stdin=diagnostics.subprocess.DEVNULL, + [sys.executable, "-B", "-c", code], stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, ) @@ -474,7 +595,7 @@ def test_unidentified_live_child_is_reaped_without_group_signals(self): def test_missing_kernel_child_authority_denies_all_signals(self): process = diagnostics.subprocess.Popen( - [sys.executable, "-c", "import time; time.sleep(60)"], + [sys.executable, "-B", "-c", "import time; time.sleep(60)"], stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, ) @@ -498,7 +619,7 @@ def test_missing_kernel_child_authority_denies_all_signals(self): def test_known_capture_without_waitid_falls_back_to_verified_child_only(self): process = diagnostics.subprocess.Popen( - [sys.executable, "-c", "import time; time.sleep(60)"], + [sys.executable, "-B", "-c", "import time; time.sleep(60)"], stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, ) @@ -523,7 +644,7 @@ def test_known_capture_without_waitid_falls_back_to_verified_child_only(self): def test_ps_zombie_anchor_rejects_reparented_wrong_group_and_malformed_rows(self): process = diagnostics.subprocess.Popen( - [sys.executable, "-c", "import time; time.sleep(60)"], + [sys.executable, "-B", "-c", "import time; time.sleep(60)"], stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, ) @@ -565,7 +686,7 @@ def test_ps_zombie_anchor_rejects_reparented_wrong_group_and_malformed_rows(self def test_cleanup_reap_lock_contention_cannot_exceed_deadline_or_signal(self): process = diagnostics.subprocess.Popen( - [sys.executable, "-c", "import time; time.sleep(60)"], + [sys.executable, "-B", "-c", "import time; time.sleep(60)"], stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, ) @@ -619,7 +740,7 @@ def close(process, **kwargs): mock.patch.object(diagnostics.subprocess, "Popen", side_effect=spawn), mock.patch.object(diagnostics, "_close_probe", side_effect=close), ): - code, output = diagnostics._probe_output([sys.executable, "-c", code], + code, output = diagnostics._probe_output([sys.executable, "-B", "-c", code], project=self.project, environment=diagnostics._environment(self.home)) self.assertEqual((code, output.strip()), (0, "codex-cli 0.159.2")) self.assertFalse(_live_group_exists(captured_processes[0].pid)) @@ -652,7 +773,7 @@ def test_unidentified_descendant_cleanup_is_unconfirmed_without_group_signals(se "os.close(write_ready)\nos.read(read_ready,1)\nos.close(read_ready)\ntime.sleep(60)\n" ) process = diagnostics.subprocess.Popen( - [sys.executable, "-c", code], stdin=diagnostics.subprocess.DEVNULL, + [sys.executable, "-B", "-c", code], stdin=diagnostics.subprocess.DEVNULL, stdout=diagnostics.subprocess.PIPE, stderr=diagnostics.subprocess.DEVNULL, start_new_session=True, ) diff --git a/test/core/test_experiment_assignment_store.py b/test/core/test_experiment_assignment_store.py index 9946bd2..2d782ce 100644 --- a/test/core/test_experiment_assignment_store.py +++ b/test/core/test_experiment_assignment_store.py @@ -14,7 +14,7 @@ from test_lifecycle import profile from devsquad.contracts import ContractError from devsquad.experiment_provenance import assignment_for, paired_input_identity -from devsquad.store import ConflictError, Store, canonical_json, git_common_dir +from devsquad.store import ConflictError, Store, canonical_json, git_common_dir, SUPPORTED_SCHEMA_VERSION class ExperimentAssignmentStoreTest(unittest.TestCase): @@ -189,7 +189,7 @@ def test_schema_13_migration_preserves_legacy_evaluation_bytes(self): saved = upgraded.connection.execute('SELECT * FROM experiments').fetchone() self.assertEqual(saved['spec_json'], spec) self.assertEqual(saved['evaluation_json'], evaluation) - self.assertEqual(upgraded.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0], 16) + self.assertEqual(upgraded.connection.execute('SELECT MAX(version) FROM schema_migrations').fetchone()[0], SUPPORTED_SCHEMA_VERSION) self.assertEqual(upgraded.connection.execute('SELECT COUNT(*) FROM experiment_assignments').fetchone()[0], 0) diff --git a/test/core/test_handoff_store.py b/test/core/test_handoff_store.py index 31a5512..cfff039 100644 --- a/test/core/test_handoff_store.py +++ b/test/core/test_handoff_store.py @@ -862,7 +862,7 @@ def test_installed_wheel_applies_schema_four_to_twelve(self): from pathlib import Path import sqlite3 import sys -from devsquad.store import Store +from devsquad.store import Store, SUPPORTED_SCHEMA_VERSION root = Path(sys.argv[1]) root.mkdir(parents=True) @@ -883,7 +883,7 @@ def test_installed_wheel_applies_schema_four_to_twelve(self): connection.close() store = Store(database, root / "artifacts") try: - assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == 16 + assert store.connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()[0] == SUPPORTED_SCHEMA_VERSION assert store.connection.execute( "SELECT 1 FROM sqlite_master WHERE type='table' AND name='handoff_submissions'" ).fetchone() diff --git a/test/core/test_learning.py b/test/core/test_learning.py index 7614c62..8aac274 100644 --- a/test/core/test_learning.py +++ b/test/core/test_learning.py @@ -20,7 +20,7 @@ validate_experiment, validate_outcome, ) -from devsquad.store import ConflictError, Store, canonical_json +from devsquad.store import ConflictError, Store, canonical_json, SUPPORTED_SCHEMA_VERSION NOW = datetime(2026, 9, 27, 16, 0, tzinfo=timezone.utc) @@ -393,7 +393,7 @@ def test_current_schema_contains_outcome_ledger(self): store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0], - 16, + SUPPORTED_SCHEMA_VERSION, ) columns = { row[1] for row in store.connection.execute("PRAGMA table_info(outcomes)") diff --git a/test/core/test_lifecycle.py b/test/core/test_lifecycle.py index e5ceb66..67dd901 100644 --- a/test/core/test_lifecycle.py +++ b/test/core/test_lifecycle.py @@ -552,7 +552,8 @@ def test_schema_twelve_contains_lifecycle_ledger(self): version = self.store.connection.execute( "SELECT MAX(version) FROM schema_migrations", ).fetchone()[0] - self.assertEqual(version, 16) + from devsquad.store import SUPPORTED_SCHEMA_VERSION + self.assertEqual(version, SUPPORTED_SCHEMA_VERSION) tables = { row[0] for row in self.store.connection.execute( "SELECT name FROM sqlite_master WHERE type='table'", diff --git a/test/core/test_mcp.py b/test/core/test_mcp.py index 3d9d555..63635ab 100644 --- a/test/core/test_mcp.py +++ b/test/core/test_mcp.py @@ -875,6 +875,19 @@ def test_plain_installed_wheel_keeps_cli_usable_without_mcp(self): integrations.stdout.strip(), "antigravity,claude-code,codex,grok", ) + schemas = subprocess.run( + [ + str(python), "-P", "-c", + "from pathlib import Path; import sys; " + "print(','.join(sorted(path.name for path in " + "(Path(sys.prefix) / 'share/devsquad/schemas').glob('*.schema.json'))))", + ], + check=True, text=True, capture_output=True, cwd=root, env=environment, + ) + self.assertEqual( + schemas.stdout.strip(), + ",".join(sorted(path.name for path in (CORE / "schemas").glob("*.schema.json"))), + ) missing = subprocess.run( [str(squad), "mcp", "serve"], text=True, capture_output=True, env=environment, ) diff --git a/test/core/test_task_entry.py b/test/core/test_task_entry.py index 52c7b99..523bbe2 100644 --- a/test/core/test_task_entry.py +++ b/test/core/test_task_entry.py @@ -833,7 +833,7 @@ def test_native_discovery_cleans_owned_child_after_provider_parent_exits(self): ) provider_code = ( "import json,os,subprocess,sys,time\n" - f"subprocess.Popen([sys.executable, '-c', {child_code!r}, {str(ready)!r}], stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n" + f"subprocess.Popen([sys.executable, '-B', '-c', {child_code!r}, {str(ready)!r}], stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n" f"while not os.path.exists({str(ready)!r}): time.sleep(.01)\n" "for line in sys.stdin:\n" " item=json.loads(line)\n" @@ -850,7 +850,7 @@ def test_native_discovery_cleans_owned_child_after_provider_parent_exits(self): def native_popen(argv, *positional, **keywords): if argv[:2] != ["/fixture/codex", "app-server"]: return real_popen(argv, *positional, **keywords) - process = real_popen([sys.executable, "-c", provider_code], *positional, **keywords) + process = real_popen([sys.executable, "-B", "-c", provider_code], *positional, **keywords) spawned["process"] = process return process from devsquad.task_entry import discover_models as real_discover_models From b83dda7fbf691503d3adf3c9ea6ecebd4071ba16 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 17:26:08 -0700 Subject: [PATCH 189/197] WIP checkpoint: release: preserve ordinary finish compatibility and pristine candidate proof (2026-10-02 17:26) --- docs/plans/engineering-team/RESUME.md | 15 +++++++++++++++ ...-terminal-readiness-partial-2026-10-02.json | 15 +++++++++++++++ plugin/core/src/devsquad/cli.py | 14 +++++++++----- test/core/test_cli.py | 18 ++++++++++++++++++ 4 files changed, 57 insertions(+), 5 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 4cdcd21..39f45a6 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -12,6 +12,21 @@ Release version0.11.0 includes core0.1.0; Council stays native-unavailable and automatic OFF. Whole-plan completion is not claimed. Root owns publication, final frozen/full/native gates and local install. No tag/release/push yet. +Release candidate `fb64e43a28e9b8eab6670be3d7da8e4fd0fcaa97` passed the +actual accepted-R5 pristine trusted check:60 tests/33.182s, full fingerprints +unchanged, zero ignored/untracked/bytecode before and after. Input fingerprint +131755f0ae8bfe75831607ffd840e1346172aa4f7452f3b2394cdd58614d5216 is NOT +a branch-review candidate identity. The exact owned clean worktree was removed +nonforce; private proof retained. First final combined full suite stopped at +45 tests/27.541s with one ordinary finish mock/call-contract failure: Council +forwarded four optionalNone kwargs. Narrow repair preserves the existing +three-argument ordinary call, forwards only opt-in Council choice, and adds +explicit forwarding coverage. Root32 CLI/Council tests/25.365s pass; +independent narrow review clean with two tests/0.012s and unchanged blobs; +Bash227/reference/diff pass. The five preflight EOF/diagnostic/fixture paths +are unchanged by this isolated CLI repair. Next freeze the repaired candidate +and run final full/native gates; retain the failed first gate truthfully. + EOF/fixture repair `4f01456189b38252a678938e7197ac2f12a6838e` is now integrated at source level: wait for natural exit without reaping under the original deadline, then owned cleanup; true version/auth status survives delayed exit. diff --git a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json index 9c3fdd3..12bb398 100644 --- a/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json +++ b/docs/plans/engineering-team/evidence/R6-terminal-readiness-partial-2026-10-02.json @@ -17,6 +17,21 @@ "093ed0f397289bffe9ea249c10796cb436027f00", "4f01456189b38252a678938e7197ac2f12a6838e" ], + "release_candidate_preflight": { + "target_revision": "fb64e43a28e9b8eab6670be3d7da8e4fd0fcaa97", + "trusted_core": "immutable accepted R5 review_worker._run_check, prepare_check_workspace, candidate_input_state imported using installed Python -P/-B", + "argv": ["env","PYTHONDONTWRITEBYTECODE=1","DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3.12","PYTHONPATH=plugin/core/src:test/core","PYTHONWARNINGS=error::ResourceWarning","python3","-m","unittest","test_diagnostics","test_task_entry"], + "tests":60,"seconds":33.182,"trusted_seconds":33.854,"returncode":0,"failures":0,"errors":0, + "input_fingerprint_before_and_after":"131755f0ae8bfe75831607ffd840e1346172aa4f7452f3b2394cdd58614d5216", + "tracked_payload_fingerprint":"3a0baeb9f87bcc8e1f1776a3b64b07cdeac1fce554ee535a87738d78f2c167bd", + "full_payload_fingerprint":"a59b2bd8e47995a93786019cc454cbf3373bf2844518f04290c7bebc5de9ddfa", + "tracked_files":333,"ignored_untracked_additions":0,"bytecode_before_and_after":0,"unchanged":true, + "fingerprint_limitation":"Input fingerprint only, NOT branch-review candidate identity", + "validated_owned_worktree_removed":"nonforce, returncode0", + "first_final_full": {"tests":45,"seconds":27.541,"failures":1,"errors":0,"unraisable":[],"reason":"ordinary finish mock/call contract received new optionalNone Council kwargs","acceptance":"failed"}, + "narrow_cli_repair": {"behavior":"Preserve3arg ordinary finish; forward any explicit Council field without inference; unchanged omittedrun resolver","root_affected":{"tests":32,"seconds":25.365,"failures":0,"errors":0},"independent_review":"clean narrow diff","independent_tests":2,"independent_seconds":0.012,"cli_blob":"ea9b622952478c423aa79d1c5d6a0f89159931ec","test_cli_blob":"698d0dda97803bc356b44084c68a5856d4c4f293","bash_assertions":227,"generated_reference":"current","diff":"passed"}, + "preflight_source_equivalence":"The five affected EOF/diagnostic/fixture paths are unchanged by the subsequent isolated CLI forwarding repair" + }, "eof_followup": { "run_id": "d2355cd0-603d-4e6a-b1a2-b05b0e99d53a", "target_revision": "8242eff68b24538aacc1b40b67b2e88e96b04d52", diff --git a/plugin/core/src/devsquad/cli.py b/plugin/core/src/devsquad/cli.py index b2e2a13..ea9b622 100644 --- a/plugin/core/src/devsquad/cli.py +++ b/plugin/core/src/devsquad/cli.py @@ -476,11 +476,15 @@ def command_resume(args: argparse.Namespace) -> tuple[dict, int]: def command_finish(args: argparse.Namespace) -> tuple[dict, int]: service = _service(args) - return envelope(data=service.finish( - _selected_run(args, service), args.disposition, args.reason, - chosen=args.choose, supported_claims=args.supported_claim, - discarded_alternatives=args.discarded_alternative, validation=args.validation, - )), 0 + run_id = _selected_run(args, service) + choice = { + "chosen": args.choose, "supported_claims": args.supported_claim, + "discarded_alternatives": args.discarded_alternative, "validation": args.validation, + } + # Ordinary R6 completion keeps its call contract; Council fields are opt-in. + if all(value is None for value in choice.values()): + return envelope(data=service.finish(run_id, args.disposition, args.reason)), 0 + return envelope(data=service.finish(run_id, args.disposition, args.reason, **choice)), 0 def command_handoff_claim(args: argparse.Namespace) -> tuple[dict, int]: diff --git a/test/core/test_cli.py b/test/core/test_cli.py index a0b176b..698d0dd 100644 --- a/test/core/test_cli.py +++ b/test/core/test_cli.py @@ -281,6 +281,24 @@ def test_finish_preserves_the_json_envelope_and_forwards_a_guided_choice(self): service.finish.assert_called_once_with("run-1", disposition, "Exact evidence assessed.") service.resolve_run_id.assert_not_called() + def test_finish_forwards_explicit_council_choice_without_inference(self): + service = mock.Mock() + response = {"run_id": "run-1", "state": "succeeded", "disposition": "accept"} + service.finish.return_value = response + code, payload, stderr = self.invoke([ + "finish", "run-1", "--accept", "--reason", "Exact evidence assessed.", + "--choose", "synthesis", "--supported-claim", "bounded retries", + "--discarded-alternative", "unbounded retries", "--validation", "checks pass", + "--runtime-dir", str(self.runtime), "--json", + ], service) + self.assertEqual((code, stderr), (0, "")) + self.assert_success_envelope(payload, response) + service.finish.assert_called_once_with( + "run-1", "accept", "Exact evidence assessed.", chosen="synthesis", + supported_claims=["bounded retries"], discarded_alternatives=["unbounded retries"], + validation="checks pass", + ) + def test_normal_doctor_default_is_readable_and_keeps_unknown_proof_unknown(self): report = {"core_version": "fixture", "ready": False, "adapters": [{ "adapter": "claude", "status": "supported", "version": "fixture", From 06767b4d112ed461683cb5c0dd0f3921a1d3fdc5 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 17:41:51 -0700 Subject: [PATCH 190/197] WIP checkpoint: release: record accepted EOF review,604-test core gate and scoped CI blockers (2026-10-02 17:41) --- docs/plans/engineering-team/RESUME.md | 17 ++++++++ .../evidence/public-release-0.11.0.json | 43 +++++++++++++++++++ 2 files changed, 60 insertions(+) create mode 100644 docs/plans/engineering-team/evidence/public-release-0.11.0.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 39f45a6..327f987 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -6,6 +6,23 @@ remain recoverable in Git; detailed receipts and failed gates stay in evidence. ## Current checkpoint — October 2, 2026 +**Latest live gates:** PR1 is open at +https://github.com/joshidikshant/devsquad/pull/1. Exact b83dda7 native EOF audit +`cb68f3ae-bea0-4744-90c4-96f25418fb8c` is succeeded22/accepted: verified +Codex0.159.2/gpt-6.1-sol/low/read_only, clean, four required checks passed with +unchanged integrity;60 affected tests/36.416s and all artifact hashes verified. +Sole frozen full core suite exec12430 on b83 PASSED604 tests/793.462s, +two optionalSDK skips, zero failures/errors/unraisable; UTC/monotonic793.88s +agree. Source/tests/scripts remained unchanged. No full core suite remains +active; do not repeat unchanged accepted core tests. Production remains +R5/schema16 pending safe local update. Public CI exposed true legacy deadline drift +(poll counts include expensive ps overhead); superseded push/PR runs cancelled +after recorded Bash failure, optionalMCP passed, core not accepted. Full-history +checkout is also needed for immutable migration fixtures. Delegated isolated +codex/release-watchdog repair/independent review is underway; integrate only +after frozen full completes. No main merge, tag or public release yet. +[Release evidence](evidence/public-release-0.11.0.json) is the current matrix. + **Public release authorized:** the user selected a public GitHub release in the existing `joshidikshant/devsquad` repository (no registry or hosted service). Release version0.11.0 includes core0.1.0; Council stays native-unavailable and diff --git a/docs/plans/engineering-team/evidence/public-release-0.11.0.json b/docs/plans/engineering-team/evidence/public-release-0.11.0.json new file mode 100644 index 0000000..22d9726 --- /dev/null +++ b/docs/plans/engineering-team/evidence/public-release-0.11.0.json @@ -0,0 +1,43 @@ +{ + "schema_version":1, + "recorded_on":"2026-10-02", + "status":"release_gates_in_progress_not_published", + "authorization":"User chose public GitHub release in existing joshidikshant/devsquad; no package registry or hosted service", + "pull_request":"https://github.com/joshidikshant/devsquad/pull/1", + "frozen_core_candidate":"b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "frozen_core_tree":"62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "native_eof_followup":{ + "run_id":"cb68f3ae-bea0-4744-90c4-96f25418fb8c", + "base":"8242eff68b24538aacc1b40b67b2e88e96b04d52", + "target":"b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "candidate_sha256":"f3243afeae0f011321d69fca1dc2f2225e4651f2a35e3452c1266cfa3da58791", + "verdict":"clean", "findings":[],"state":"succeeded","version":22,"host_disposition":"accept", + "identity":{"harness":"codex","harness_version":"codex-cli0.159.2","model_id":"gpt-6.1-sol","effort":"low","verification":"verified","permission_policy":"read_only"}, + "checks":{"diff":"passed","bash":"passed","affected_tests":60,"seconds":36.416,"reference":"passed","all_required":true,"all_integrity":"verified_unchanged"}, + "artifact_hashes_verified":["0dcb6d16b34c4f9c6e3eed6aa7e0161bf1d46c5f7f61370a7f8b80008dae5b69","46b3f1b3c8c1d8a7e824e4fef4502eb88841f27e1deaddf1b166359937f97283","d0f0b5b3e2169e34ed4ae63eacc7cdedce82b68fd9de4a904d01323f6910253f","182acd24afa59bc37c539f6fa8f0e6b1dd01275cfcd784970c2ffd8e21f00e32"], + "usage":{"source":"native_reported","input_tokens":95453,"output_tokens":517,"total_tokens":95970}, + "limits":"Narrow EOF/fixture audit only, not whole-plan or native Council acceptance; usage is not invoice/Plus-window estimate" + }, + "first_ci":{ + "push_run":37082086201,"pr_run":37082130929, + "status":"cancelled_superseded_after_confirmed_legacy_failure", + "legacy":"Actual Bash3.2 deadline failure: poll-count timing stretched1s to4s and3s to6.123s/8.644s", + "optional_mcp":"passed_both_runs", + "core":"cancelled_not_accepted; shallow checkout cannot provide immutable f4fa657/bf3 migration fixtures", + "repair":"Isolated actual sleep deadline and unchanged timing bounds; full-history core checkout and tracked spawn-safe failfast runner; independent review pending", + "no_passing_failure_claim":true + }, + "fresh_install_b83":{ + "actual_core_only":true,"version":"squad0.1.0","changed":true,"reinstall_changed":false,"status_changed":false,"all_drift":false,"manifest_matches":true, + "digest":"303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52", + "wheel_schemas":11,"wheel_migrations":17,"all_packaged_bytes_match_source":true, + "proof_sha256":"a0c7914ad5228276834ac242602fd510601f65562e406dd3dd75b31d107931ed", + "initial_failed_gate":"Whole-production filesystem equality invalidated by separately authorized concurrent root native ledger activity; perfield attribution unavailable, no unchanged-ledger claim", + "subsequent_validation":"Stable selector/manifest/installstate/launcher observed unchanged; unique private install/runtime/bin only", + "provider_or_network_calls":0,"production_update":false + }, + "final_full":{"target":"b83dda7fbf691503d3adf3c9ea6ecebd4071ba16","exec_session":12430,"tests":604,"seconds":793.462,"monotonic_seconds":793.8633149590023,"utc_seconds":793.881373,"failures":0,"errors":0,"optional_sdk_skips":2,"unraisable":[],"status":"passed","source_tests_scripts_frozen":true}, + "installation":"production remains accepted R5/schema16 until full/repair gates", + "limitations":{"council":"native-unavailable, automaticOFF, fixture comparison inconclusive; no more network attempts","jev":"OFF; authorized single pilot spent","claude_desktop_code_tab":"unverified/TCC, not substituted by CLI proof","antigravity":"agyCLI1.2.14 final-release status pending; IDE out of scope","whole_plan":"not complete"}, + "remaining":"Complete frozen core suite; integrate independently reviewed legacy deadline/CI repair; safe actual installation/SDK/agyCLI; final green public CI; merge/tag/release verified assets" +} From 765ab2da805253d8ff365d60f8c43fbad6598119 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 17:53:42 -0700 Subject: [PATCH 191/197] WIP checkpoint: release: preserve installed17,9SDK and actual updated agyCLI proof (2026-10-02 17:53) --- docs/RUNTIME-GUIDE.md | 6 +++++ docs/plans/engineering-team/RESUME.md | 12 ++++++++++ .../evidence/public-release-0.11.0.json | 23 ++++++++++++++++--- 3 files changed, 38 insertions(+), 3 deletions(-) diff --git a/docs/RUNTIME-GUIDE.md b/docs/RUNTIME-GUIDE.md index 6bf3a83..f959557 100644 --- a/docs/RUNTIME-GUIDE.md +++ b/docs/RUNTIME-GUIDE.md @@ -59,6 +59,12 @@ releases and saved receipts remain present. Custom runtime directories get the same migration guard on first access; use `DEVSQUAD_RUNTIME_DIR` for the installer's explicitly scoped ledger check. +A long-running MCP server keeps the package it started with. After a schema +upgrade, reconnect only the DevSquad MCP connection in each already-open host +to load the selected release. An old server may return `SCHEMA_UNSUPPORTED`; +that is the old-client fence, not lost work. Do not remove the ledger or change +unrelated servers. New CLI/MCP processes already use the stable launcher. + Antigravity's non-interactive print mode also enforces project permissions. For unattended read-only status checks, add this exact grant to the DevSquad project's Permissions list in Antigravity: diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 327f987..deaccff 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -23,6 +23,18 @@ codex/release-watchdog repair/independent review is underway; integrate only after frozen full completes. No main merge, tag or public release yet. [Release evidence](evidence/public-release-0.11.0.json) is the current matrix. +Newer local gate: immutable core release0.1.0-py31214-303a0e0a6c87-mcp-a26bc88afbef +is installed, schema17, source303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52. +Zero nonterminal before/after; pre16 online backup0600/integrityOK retained. +Reinstallchangedfalse/all driftfalse/manifests match/pipcheck pass; nine actual +installed SDK tests/3.126s, no skips/errors/failures. Doctor ready for review +and Claude→Codex delivery; four registrations match. Grok/agy worker-readiness +stays unknown, Councilnativefalse. Saved cb68 still succeeded22. +Updated agy1.2.14/Gemini3.8FlashLow actually called DevSquad status once and +observed cb68 succeeded22 in18.654s. It read its own MCP schema and generated +tool-result file, not project files/other servers; this is not a zero-file-read +claim, IDE proof or automatic worker coverage. No IDE settings were changed. + **Public release authorized:** the user selected a public GitHub release in the existing `joshidikshant/devsquad` repository (no registry or hosted service). Release version0.11.0 includes core0.1.0; Council stays native-unavailable and diff --git a/docs/plans/engineering-team/evidence/public-release-0.11.0.json b/docs/plans/engineering-team/evidence/public-release-0.11.0.json index 22d9726..a56dbc0 100644 --- a/docs/plans/engineering-team/evidence/public-release-0.11.0.json +++ b/docs/plans/engineering-team/evidence/public-release-0.11.0.json @@ -37,7 +37,24 @@ "provider_or_network_calls":0,"production_update":false }, "final_full":{"target":"b83dda7fbf691503d3adf3c9ea6ecebd4071ba16","exec_session":12430,"tests":604,"seconds":793.462,"monotonic_seconds":793.8633149590023,"utc_seconds":793.881373,"failures":0,"errors":0,"optional_sdk_skips":2,"unraisable":[],"status":"passed","source_tests_scripts_frozen":true}, - "installation":"production remains accepted R5/schema16 until full/repair gates", - "limitations":{"council":"native-unavailable, automaticOFF, fixture comparison inconclusive; no more network attempts","jev":"OFF; authorized single pilot spent","claude_desktop_code_tab":"unverified/TCC, not substituted by CLI proof","antigravity":"agyCLI1.2.14 final-release status pending; IDE out of scope","whole_plan":"not complete"}, - "remaining":"Complete frozen core suite; integrate independently reviewed legacy deadline/CI repair; safe actual installation/SDK/agyCLI; final green public CI; merge/tag/release verified assets" + "installation":{ + "release":"0.1.0-py31214-303a0e0a6c87-mcp-a26bc88afbef","source_digest":"303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52", + "python":"3.12.14","mcp":"2.2.0","ledger_schema":17,"nonterminal_runs_before_and_after":0,"private_pre16_online_backup":"retained0600_integrity_ok", + "first_changed":true,"reinstall_changed":false,"all_drift":false,"manifest_matches":true,"pip_check":"passed","previous_releases_retained":true,"downloads":0, + "installed_sdk":{"tests":9,"seconds":3.126,"failures":0,"errors":0,"skips":0,"origin":"selected immutable release/core/src, not primary checkout"}, + "sdk_harness_first_assertion":"Initial helper assumed site-packages origin and aborted before tests; installer intentionally imports its immutable copied core/src. Corrected origin assertion retains release-bound/no-primary-source check; nine tests then passed.", + "doctor":{"ready":true,"branch_review_ready":true,"issue_delivery_ready":true,"matching_registrations":4,"claude_subscription_authenticated":true,"codex_subscription_authenticated":true,"council_native_ready":false,"grok_antigravity_worker_readiness":"unverified/unknown, not implied by MCP registration"}, + "saved_run_continuity":{"run_id":"cb68f3ae-bea0-4744-90c4-96f25418fb8c","state":"succeeded","version":22} + }, + "updated_antigravity_cli":{ + "version":"1.2.14","observed_init_model":"gemini-3.8-flash-low","mode":"plan/sandbox; native request-review","operation":"one devsquad/squad_status call","run_id":"cb68f3ae-bea0-4744-90c4-96f25418fb8c","observed_state":"succeeded","observed_version":22, + "seconds":18.65435924999838,"returncode":0,"timed_out":false,"native_result":"SUCCESS", + "stdout_sha256":"cc6045878de2124b845adab33401ecd1f52317dd1f6e654b2b439041b2da1863","stderr_sha256":"3a7f8550106d86abb322e8ed0fb30a0296c0e97c6be03d5a8033b99d16737459", + "actual_mcp_output_sha256":"d22dccedbcf001999c59d6167820e04a833484f70a2d738e2e7f5c708b9a98f8", + "native_file_reads":"Own DevSquad MCP schema and provider-generated tool result file only; not zero file reads. Actual tool-result envelope independently parsed and matched saved run; no project files, other servers or mutations.", + "limits":"CLI MCP operation proof only; not Antigravity IDE, automatic writer/reviewer, cost savings or whole-plan acceptance" + }, + "codex_app_existing_connection":{"operation":"squad_status cb68","result":"SCHEMA_UNSUPPORTED","reason":"existing long-running server is bound to old package; reconnect only DevSquad connection to select new release","new_cli_and_agy_same_run":"succeeded22","user_action":"async reconnect request pending; no global restart/settings changes"}, + "limitations":{"council":"native-unavailable, automaticOFF, fixture comparison inconclusive; no more network attempts","jev":"OFF; authorized single pilot spent","claude_desktop_code_tab":"unverified/TCC, not substituted by CLI proof","antigravity":"agyCLI1.2.14 updated-release MCP status verified; automatic worker roles unverified and IDE out of scope","whole_plan":"not complete"}, + "remaining":"Integrate independently reviewed legacy deadline/CI repair; final green public CI; merge/tag/release verified assets. Existing CodexApp connection needs scoped reconnect; core604/native/installed9SDK/agyCLI gates already passed, do not repeat unchanged." } From 5a576d39469c198a3cc93474960886b875363419 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 18:06:21 -0700 Subject: [PATCH 192/197] WIP checkpoint: release: integrate independently accepted portable watchdog and preserve current gates (2026-10-02 18:06) --- .github/workflows/offline.yml | 4 +- docs/plans/engineering-team/RESUME.md | 587 +++--------------- docs/plans/engineering-team/backlog.json | 32 +- .../legacy-watchdog-release-repair.json | 111 ++++ .../evidence/public-release-0.11.0.json | 248 ++++++-- plugin/lib/adapter.sh | 93 ++- test/test_m1_legacy.sh | 222 ++++++- 7 files changed, 701 insertions(+), 596 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/legacy-watchdog-release-repair.json diff --git a/.github/workflows/offline.yml b/.github/workflows/offline.yml index f1e41ad..2d23111 100644 --- a/.github/workflows/offline.yml +++ b/.github/workflows/offline.yml @@ -28,6 +28,8 @@ jobs: python: ["3.11", "3.14"] steps: - uses: actions/checkout@v4 + with: + fetch-depth: 0 - uses: actions/setup-python@v5 with: python-version: ${{ matrix.python }} @@ -41,7 +43,7 @@ jobs: PYTHONDONTWRITEBYTECODE: "1" PYTHONPATH: plugin/core/src PYTHONWARNINGS: error::ResourceWarning - run: python -m unittest discover -s test/core -q + run: python scripts/run-core-tests.py --failfast optional-mcp: name: Optional MCP 2.2.0 (provider-offline) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index deaccff..eecb0c4 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -1,500 +1,97 @@ # Resume DevSquad after an interruption -This is the authoritative recovery entry point, not a chronological chat log. -Compare it with Git status/recent commits and retain newer work. Earlier notes -remain recoverable in Git; detailed receipts and failed gates stay in evidence. +Read this checkpoint, backlog.json and SOL-HANDOFF.md, then compare Git status +and recent commits. Retain newer work. Continue codex/engineering-team; no stash, +reset or restart from main. Detailed receipts and failed gates remain in evidence. ## Current checkpoint — October 2, 2026 -**Latest live gates:** PR1 is open at -https://github.com/joshidikshant/devsquad/pull/1. Exact b83dda7 native EOF audit -`cb68f3ae-bea0-4744-90c4-96f25418fb8c` is succeeded22/accepted: verified -Codex0.159.2/gpt-6.1-sol/low/read_only, clean, four required checks passed with -unchanged integrity;60 affected tests/36.416s and all artifact hashes verified. -Sole frozen full core suite exec12430 on b83 PASSED604 tests/793.462s, -two optionalSDK skips, zero failures/errors/unraisable; UTC/monotonic793.88s -agree. Source/tests/scripts remained unchanged. No full core suite remains -active; do not repeat unchanged accepted core tests. Production remains -R5/schema16 pending safe local update. Public CI exposed true legacy deadline drift -(poll counts include expensive ps overhead); superseded push/PR runs cancelled -after recorded Bash failure, optionalMCP passed, core not accepted. Full-history -checkout is also needed for immutable migration fixtures. Delegated isolated -codex/release-watchdog repair/independent review is underway; integrate only -after frozen full completes. No main merge, tag or public release yet. -[Release evidence](evidence/public-release-0.11.0.json) is the current matrix. - -Newer local gate: immutable core release0.1.0-py31214-303a0e0a6c87-mcp-a26bc88afbef -is installed, schema17, source303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52. -Zero nonterminal before/after; pre16 online backup0600/integrityOK retained. -Reinstallchangedfalse/all driftfalse/manifests match/pipcheck pass; nine actual -installed SDK tests/3.126s, no skips/errors/failures. Doctor ready for review -and Claude→Codex delivery; four registrations match. Grok/agy worker-readiness -stays unknown, Councilnativefalse. Saved cb68 still succeeded22. -Updated agy1.2.14/Gemini3.8FlashLow actually called DevSquad status once and -observed cb68 succeeded22 in18.654s. It read its own MCP schema and generated -tool-result file, not project files/other servers; this is not a zero-file-read -claim, IDE proof or automatic worker coverage. No IDE settings were changed. - -**Public release authorized:** the user selected a public GitHub release in -the existing `joshidikshant/devsquad` repository (no registry or hosted service). -Release version0.11.0 includes core0.1.0; Council stays native-unavailable and -automatic OFF. Whole-plan completion is not claimed. Root owns publication, -final frozen/full/native gates and local install. No tag/release/push yet. - -Release candidate `fb64e43a28e9b8eab6670be3d7da8e4fd0fcaa97` passed the -actual accepted-R5 pristine trusted check:60 tests/33.182s, full fingerprints -unchanged, zero ignored/untracked/bytecode before and after. Input fingerprint -131755f0ae8bfe75831607ffd840e1346172aa4f7452f3b2394cdd58614d5216 is NOT -a branch-review candidate identity. The exact owned clean worktree was removed -nonforce; private proof retained. First final combined full suite stopped at -45 tests/27.541s with one ordinary finish mock/call-contract failure: Council -forwarded four optionalNone kwargs. Narrow repair preserves the existing -three-argument ordinary call, forwards only opt-in Council choice, and adds -explicit forwarding coverage. Root32 CLI/Council tests/25.365s pass; -independent narrow review clean with two tests/0.012s and unchanged blobs; -Bash227/reference/diff pass. The five preflight EOF/diagnostic/fixture paths -are unchanged by this isolated CLI repair. Next freeze the repaired candidate -and run final full/native gates; retain the failed first gate truthfully. - -EOF/fixture repair `4f01456189b38252a678938e7197ac2f12a6838e` is now integrated -at source level: wait for natural exit without reaping under the original -deadline, then owned cleanup; true version/auth status survives delayed exit. -Hung EOF remains bounded/unknown. Controlled Python fixtures use `-B`; native -provider environment and integrity gates are unchanged. Agent strict40 tests -pass on Python3.12/12.894s and3.14/13.398s; root34 diagnostics/11.159s and -eight reconciliation/SDK checks/21.651s pass. Wheel completeness regression -failed as expected (one/2.657s: Council schema missing); packaging fix passes -the same actual installed-wheel test (one/2.831s), no SDK required. - -Final nongenerating Council diagnostic is exhausted: eight denial controls -pass; sandbox backend RPC fails network_request_failed/noHTTP, unsandboxed -control fails strict response validation. Zero generating calls, auth copy -removed and original unchanged. No more network attempts or permission widening. -[Portable evidence](evidence/r7-final-nongenerating-network-diagnostic.json). -Next: checkpoint the coherent integration after Bash/reference gates, prove -the exact trusted check in a pristine candidate worktree, then one narrow EOF -native review and one final frozen combined full suite. Safely update the -local installation and recheck agy CLI; publish only the tested candidate. - -Council reconciliation `0e7dd319b56802329187007738970e26b5609b1d` is integrated, -not accepted/installed. It includes preserved42979ba/72ae413/eee6db7, schema17, -R6 canonical terminal authority, explicit choice/copyable recovery, headless -crash/expiry recovery and shared bootstrap cleanup. Its168 affected tests/ -148.693s (two SDK skips), two actual SDK/2.964s, Bash/reference/diff pass. -Original failed169 loader command and test-only inactive-claim errors are saved -in [reconciliation evidence](evidence/r7-council-reconciliation-partial.json). -Native network remains unavailable before generation; automatic off/comparison -inconclusive. Ownership093 and EOF4f are integrated as described above. -Root owns final frozen full/native/install gates; production remains R5/schema16. - -Accepted work must not be restarted or re-tested unchanged: - -- R3 `39b95f1`: exact native review clean, focused/public/historical evidence - and source-equivalent prior477 full pass. [Closure](evidence/R3-closure-2026-10-02.json). -- R4 `913abc1`: normal catalogs, account-wide quota fences and guarded aliases; - repaired exact native audit accepted,484 full tests, installed SDK9 pass. - [Closure](evidence/R4-closure-2026-10-02.json). -- **Current installed R5 `dfe9976`, schema16**: public outcomes/trials and - shared experiment budgets; repair audit `e046a174-55ce-4496-973e-2fc97c68a493` - accepted,504 full tests/511.241s, installed SDK9/2.674s, no drift and four - registrations matching. [Closure](evidence/R5-closure-2026-10-02.json). - -Failed/red/interrupted gates remain in each package's partial evidence. The -remaining delivery is R6 → R7/C1 → final R8 acceptance, not architecture work. -Antigravity is **agy CLI**, now1.2.14; its final-release operation is pending. -The older1.2.13 receipt is historical, not proof of the updated CLI/adapter. - -R6 terminal/readiness checkpoints `dd709804`, `79f9b767`, `b785d040`, -`d0eec55c`, shared helper `4a6bc15d` and catalog reuse `03124c79` are now -integrated for the R6 source checkpoint in the main tree. -The fresh temporary install reaches review/fix receipts with only provider -binaries faked: actual normal discovery, distinct verified Claude/Codex -identity, required seeded Python check, guided finish, original checkout -unchanged, and project-scoped zero/one/multiple run choices. No task/decision -JSON or internal Service.start fixture is used. Agent gates and failures are -in `evidence/R6-terminal-readiness-partial-2026-10-02.json`; this is not yet -an accepted or installed R6 package. Root's combined focused invocation had -two nonexistent module names: the loader errors are not a passing gate or -product failures. Corrected affected/full gates remain required. - -Readiness `79f9b767` fixes the root-reviewed exited-parent/ignoring-descendant -cleanup defect. The same red reproduction in native catalog discovery is -repaired by the shared bounded helper; `03124c79` has 106 affected tests/ -95.408s with no skips/warnings, Bash and reference gates. Root's integrated -114-test gate (two optional SDK skips), 32 delivery/handoff tests and final -27 task-entry/fresh-install tests in 25.389s pass. Independent terminal review -found a real P1: interruption after guided finish acquires its claim but before -completion leaves no saved claim for retry; initial-only refuses it even after -expiry. A controlled offline reproduction confirms this, not a usage timeout. -Repair checkpoint `22a4ed8a` is now integrated: the canonical decision and exact -packet hash commit atomically with the claim. Live identical retries preserve -the claim; expired identical retries acquire a fresh fence only with the latest -matching durable terminal marker. App claims (including the same owner name), -changed decisions, corruption, cancellation and stale/new handoff races fail -closed. Saved submissions resume normally; Next commands are copyable. Agent -gates: 150 affected tests/180.750s and final 45 CLI/UX tests/48.316s; Bash 227, -reference and diff checks pass. Its first expanded gate exposed only an old -isolated bf3 schema-15 test expectation; root already expected 16. Historical -schema-four fixture checks are retained with the current supported-version -assertion. Root's integrated 61 terminal/CLI/store/handoff tests pass in -55.963s with ResourceWarning strict. Independent follow-up passed ten existing -repair tests/26.811s but found two edge cases: an `expired_claim` rejection -poisons identical retry even after its own fresh fence, and a reason beginning -with `--` makes the emitted separate `--reason` argument invalid. The terminal -agent repaired these at `31a8a68b`: complete rejected-row/digest immutable audit -and exact fresh-authority recovery, consecutive event versions, unchanged app -replay and actual `--reason=` CLI round-trip. Its 156-test affected gate passes -in 350.005s, no skips/warnings; independent six-test/9.120s review is clean and -exact reviewed blobs/diff match the checkpoint. Final `e7e9042c` additionally -requires the canonical original rejection event to match its row/run/version/ -timestamp/ID/hash/reason; 19 focused tests/19.207s pass, 13 negative subcases. -Both follow-ups are integrated; root's final 19 focused tests pass in 7.183s. -Bash/reference/diff gates pass. Final added-hunk independent review is clean; -one test/13 negative subcases in 0.208s, reviewed/committed blobs equivalent. -Exact immutable R6 candidate: `7130ee7689f8cc14efacb606bcbc4c05f373298c`. -Native independent audit `44d0eecb-b5da-4c09-b1e1-c38b0d1d09ca` is explicitly -**rejected**, failed/version22. Verified Codex0.159.2/gpt-6.1-sol/low found P1 -`r6-probe-ownership-unavailable`: missing captured/current start identities -permit group signals and diagnostic protocol/output use. Repair `093ed0f397289bffe9ea249c10796cb436027f00` -is now integrated: missing capture blocks reads/peer/send and all group signals; -only kernel-confirmed direct children are stopped/reaped. Surviving descendants -without authority remain cleanup-unconfirmed. Known missing-current identity -requires a retained unreaped waitid or exact bounded zombie PID/PPID/PGID anchor -under the Popen reap lock; fake/reaped/live-ps/wrong-parent/wrong-group rows fail -closed. EOF preserves the child PID through cleanup. No ABI additions; helper -requires exclusive Popen reaping, never an external waitpid/SIGCHLD reaper. -Six red baseline regressions/1.216s; accepted Python3.12 strict36/11.250s and -Python3.14 strict36/11.364s pass; root seven/1.259s with exact matching blobs, -Bash/reference/diff green. Diff/Bash -checks passed, but the required66-test command (89.977s/OK) was **invalidated** -by undeclared bytecode; reference was not run. The private invocation now sets -PYTHONDONTWRITEBYTECODE=1; candidate integrity was not relaxed. Exact packet -and four artifact hashes are verified and retained in R6 partial evidence. -The one frozen baseline full gate completed successfully, exec86382: -`env DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3.12 PYTHONPATH=plugin/core/src:test/core PYTHONWARNINGS=error::ResourceWarning python3 scripts/run-core-tests.py --failfast`. -**554 tests/645.034s**, two optional SDK skips, zero failures/errors/unraisable; -UTC/monotonic645.167s agree. RootHEAD32b94d6 source/tests/scripts are equivalent -to7130ee7. This is a baseline pass, not proof of the later ownership repair. -Follow-up `d2355cd0-603d-4e6a-b1a2-b05b0e99d53a` reviewed exact root8242eff, -**rejected**, failed/version22: medium `R6-EOF-exit-status`. Closing stdout is -not natural exit; immediate cleanup TERM can turn valid version/authentication -into unknown. Readiness repairs bounded nonreaping completion before cleanup. -Its56 tests/27.584s ran OK, but integrity again invalidated source bytecode: -controlled provider fixture children intentionally drop Python env flags and -import the core. Fix those fixture children, not native environment or integrity. -Before another native audit, run the exact trusted check in a pristine offline -candidate worktree and verify the full input fingerprint is unchanged. -Resolver-socket-only nongenerating diagnostic failed actual rate-limits RPC -(-32603/network_request_failed, no HTTP response); all eight access/child probes -and owned cleanup passed, temporary auth copy removed. No generating request. -Council agent owns one paired bounded narrow-DNS diagnostic, never broad access. -No full/native run remains active. Next save integrated Council after root -affected/Bash gate, integrate the EOF/fixture repair, then pristine check proof, -narrow exact native follow-up and one final frozen combined full gate. Install only after -final frozen/native acceptance. Production remains accepted R5/schema16. -R7 Council remains in isolated `r7-council`; its controlled stage flow is -partial. Its source is checkpointed at `42979ba4` with 21 Council tests and -67 shared-contract tests (two optional SDK skips), Bash/reference/diff gates. -Default-deny macOS own-evidence/peer-ledger denial and native bootstrap/catalog -probes are boundary mechanics, not a live Council receipt. Native HTTPS, -reserved-launch cancellation, comparison and final acceptance remain open. -A controlled public-start/actual-worker scheduling-barrier reproduction proves -the immutable installed R5/schema16 client can acquire a Council headless -handoff that the new client rejects, stranding its completed lead. Both stores -were authoritatively schema16; temporary runs were cancelled and owned worker -PIDs confirmed absent. The Council agent owns a minimal schema17 compatibility -epoch and old-client/active-upgrade tests. Residual checkpoint `72ae4130` now -has the minimal no-table schema17 epoch, owned gated-launch cancellation and -actual matched/held-out fixture workflow comparison: inconclusive, all native -quality/escaped-defect/rework/quota/host-usage observations remain unknown, -automatic use off. Its 25 Council tests/90.724s, 35 shared migration tests/ -54.469s, 12 handoff-store tests/4.528s and 10 supervisor/crash tests/1.819s pass. -Native HTTPS attestation fails closed before generation. The Council agent -completed the actual temporary immutable16-to17 install-upgrade proof at -`eee6db738a9996ea432e3ca2bf2347b005ecd74a`: active and recoverable16 both defer -without selector/launcher/schema changes; safe cancel/reconcile permits17, -then immutable long-lived/fresh old16 clients deny Council mutations. Two -focused tests/7.976s, Bash/reference/diff pass; no production update occurred. -Readiness independently verified old16 denial on17 and the actual fourth lead -succeeding. Shared-hunk/formatter/R6 authority reconciliation0e7dd3 is integrated -as described above; original isolated branches are retained. Root remains on -codex/engineering-team. No native Council -generation occurred. Preserve all original checkpoints/attached worktrees. - -Desktop control worked for scoped inspection. Claude's local Code tab selected only this -DevSquad project on `codex/engineering-team`, with an empty prompt; no proof -request was sent. The user clarified **Antigravity means CLI, not IDE**; -do not treat IDE trust/UI as a required Gemini acceptance gate. An IDE folder -chooser was inspected only: project trust, settings and MCP servers were not -changed. The cancellation attempt returned a new TCC capture denial, so UI -closure is unverified; do not bypass it. Recheck updated agy 1.2.14 CLI against -the accepted final release. Unrelated servers/settings remain untouched. - -Workspace: /Users/Dikshant/Desktop/Projects/devsquad. -Branch: `codex/engineering-team`; never restart this build from main. -Historical runtime repairs and the accepted G4 candidate below were integrated -and installed at `6d2e0ba` (superseded by the current R5 release); all five blob hashes -exactly match the accepted candidate. Local affected gate: 27 tests in 18.940s, -plus 227 Bash assertions, generated-reference and diff checks passed. - -### Latest verified result - -Genuine saved issue delivery **`360a4993-d016-406e-8bb4-ae6d57dfe6cf`** is -**succeeded/version 31**, with fenced host acceptance: - -- Exact candidate: `f8c4f8c83f170870eb37d564e54eee6188fc233c`. -- Candidate SHA-256: `3f0fa0e86edd8a4c7b04bc1a2447d70b136e596cafe3624826f1f46c2fcee568`. -- Verified native Claude Sonnet 5 writer (Claude 2.1.220); independent verified - Codex gpt-6.1-sol/low read-only reviewer, clean, no findings. -- All three mandatory checks passed with unchanged verified integrity: diff, - 227 Bash assertions, **477 core tests in 455.860s**, two optional-SDK skips, - zero errors/failures/unraisable diagnostics. UTC/monotonic ~456.22s agree. -- Explicit offline DEVSQUAD_BUILD_PYTHON made installed-wheel gates execute. -- Real Git README/symlink/Unicode/byte-documentation/byte-test cases all pass. - Byte-name fixtures use index plumbing because APFS rejects those names. - -The final **complete-diff branch review** is **succeeded/version 22**, accepted: -**`a6ecd887-5122-4281-b988-4d344a662e20`**, exact base -`1781b89b1228dfca2ca9148df437af3ac04b971f` → candidate `f8c4f8c…`. -Its scope is only the five changed files, with mandatory diff/Bash/affected -tests; no duplicate full gate. All checks passed with unchanged verified -integrity; 27 affected tests in 24.173s. Verified native Codex review is clean. - -### Exact next action - -1. The user's three requested runtime actions are verified. Do not repeat - these accepted proofs or the unchanged full suite. Both exact packets and - artifact hashes are verified; never edit frozen evidence. -2. R3/R4/R5 are accepted by their October 2 closure matrices. Finish integrated - R6 UX/readiness and its shared probe cleanup, then R7/C1 and R8 whole-delivery - acceptance. Preserve the existing reader and - lifecycle eligibility gates; do not restart the architecture exercise. -3. Claude Code-tab proof remains separate and unverified; current control - returned TCC denial, do not bypass it. Antigravity IDE is explicitly outside - the user's clarified request; Gemini acceptance uses the agy CLI. Grok/Gemini - MCP status calls do not prove automatic writer/reviewer adapters. - -Only one full core suite may run at a time; freeze source/tests while it runs. -Use the tracked spawn-safe scripts/run-core-tests.py, never a stdin main. -Run bash test/run.sh to completion before every commit. Keep a clean tree; -use git-safety checkpoints, never stash. No goal is currently active. - -## Current local installation and host proof - -Stable launcher: /Users/Dikshant/.local/bin/squad. -Selected release: `0.1.0-py31214-90f1e87fb9ac-mcp-a26bc88afbef`. -Python 3.12.14/MCP 2.2.0, schema 16; previous releases and private pre-upgrade -SQLite backup retained. Source/plugin/installed payload drift is false. -Upgrade defers for old active/recoverable runs, swaps without migration under -the lock, then lazily migrates with old-client write guards. - -- Real Claude MCP lead claimed/renewed/completed - `b9501b55-1c65-4b32-a03f-1087cca8fefc`, succeeded/version 23. - This is CLI/portable handoff, not desktop local Code-tab UI proof. -- Grok user-approved normal updater installed 1.0.46 stable; old executable - backed up privately, login/settings preserved. Native grok-4.7-build actually - called DevSquad status. This is not automatic Grok writer/reviewer proof. -- Antigravity 1.2.13/Gemini 3.8 Flash Low actually called the same MCP status. - Final-install recheck actually observed accepted run 360a4993, succeeded/31, - in 12.118s. Existing project-only status grant, plan/sandbox, no bypass. - IDE UI permission was denied; do not bypass it. -- Initial installed SDK gate: 22 passed/no skips; final installed transport - gate: 9 passed in 2.803s, no skips. Reinstall unchanged, no payload drift, - pip check/doctor passed, all four host registrations ready/unchanged. - -## Repairs and failure history to preserve - -Full portable redacted record: -[evidence/R8-installed-workflows-2026-10-01.json](evidence/R8-installed-workflows-2026-10-01.json). - -- Claude argv variadic tools consumed the prompt: explicit -- separator fixed. -- Strict bounded native JSON/stream identity verifies one session-correlated - writer model, retaining auxiliary usage. firstParty maps to Anthropic only - for verified exact Claude 2.1.220; unknown serving revisions stay unknown. -- Detached login needs HOME plus USER. Minimal environment and HOME-only - reproduced synthetic not-logged-in output. Allowlist repaired; no API keys - or provider overrides inherited; native auth errors fail without identity. -- `45667697…` writer succeeded, but reviewer runner/child died without a - receipt; cause unknown. Dead ownership safely cancelled/version 17. -- `288ee6f9…` clean native review, but 471-test gate had three missing - build-interpreter helper errors. Host rejected, failed/version 31. -- `33c64397…` first iteration passed real 475-test workflow, but exact - combined review `ba14b743…` found strict UTF-8 byte-name decoding; rejected. - Its fenced Claude revision ignored the new issue and made no edits. Prompt - hash/reason were validated; correctly failed/version 37 for no changes. - It is **not** a quota timeout or accepted candidate. Fresh task above fixed it. -- G4 now uses exact target Git regular blobs/trees, actual tracked test*.py, - Bash plus core checks, required checks, exact-argv dedup and byte/NUL-safe - filename parsing. Three wheel-test helpers catch unavailable interpreter - candidates without relaxing isolation/dependency checks. - -## Broader plan / constraints - -M1/M2 remain accepted. R1/R2 source/offline repairs are verified; R3 saved-run -reader, eligibility/revisions, provenance/ratio and upgrade repairs are now -accepted with the independent R3 closure proof above. -Normal aliases/catalog/quota are accepted by the R4 closure above. This does -not close R5 public trials/outcomes, remaining R6 UX or R7/C1 Council. -Read backlog.json and SOL-REVIEW-FOLLOWUP.md for dependency/acceptance details; -do not restart the architecture exercise or weaken gates to mark these done. - -Jev .env is private/ignored and complete. Exactly one authorized pilot request -was spent; 8/8 family and skill labels, 6/8 tier labels including a confident -wrong tier. Runtime classification remains OFF; no adoption benefit or Laya -setup is proven. Do not repeat the pilot or silently use hosted/API fallback. - -No credit purchases/resets, paid API fallback, global AI settings or unrelated -external messages are authorized. The user's latest GitHub-release choice -authorizes scoped push/PR/merge/tag/release in joshidikshant/devsquad, superseding -the earlier publication prohibition for that destination only. -Raw provider output and credentials -stay outside Git. Native reported cost/usage is not an inspected subscription -invoice or a reliable number of Plus five-hour windows. - -Privacy: an earlier global Antigravity MCP listing exposed an unrelated -StitchMCP credential; user was advised to rotate it. Never repeat/store it. -Filter future diagnostics to DevSquad only; do not change unrelated credentials. - -Private bounded helpers are under -/Users/Dikshant/.devsquad/private-probes/r8-installed-workflows-20261001: -observe-run.py, inspect-handoff.py, complete-handoff.py, start-combined-review.py, -g4-candidate-cases.py and probe.py. Claims/raw logs remain private; never reuse -the completed b950 handoff's prior claim for a different run. - -## Isolated R7 source checkpoint — October 2, 2026 - -The Council slice is prepared in the delegated `r7-council` worktree based on -`bf3de086`, not installed or accepted as native production functionality. It -reuses existing detached ownership, attempt/account reservations, immutable -artifacts, handoff fences and terminal reporting: proposer A, proposer B, -critic, then the existing sole host/headless lead. No daemon, writer or general -scheduler was added. The submitted snapshot remains frozen separately from -fenced stage projections. Automatic triggering is always OFF; one round is -supported (`max_revisions=0`). Another round needs a new explicitly capped run. - -The normal surface is `squad council QUESTION --read-path PATH --max-invocations -4 --dry-run`; three distinct entitled Codex model IDs may share the subscription -harness. Host completion has `council-finish` with an explicit chosen label or -synthesis, claims, discarded alternatives and validation; no manual JSON is -required. Root must reconcile these narrow CLI/service/store hunks with R6's -normal formatter and atomic guided-finish claim marker before integration. - -Actual same-user macOS Seatbelt probes demonstrate default-deny own-evidence -reads while peer evidence, ledger, private logs, artifacts, symlink escapes and -child processes remain denied. Unsupported OS has no unsafe fallback. Native -Codex isolated initialization/catalog listing was non-generating only. Narrow -CFPreferences service/shared-memory support works; HTTPS model refresh still -fails under diagnostic DNS additions. Native subscription/network readiness -and generating receipts are **not proven**; no generating calls were made. -Fixture receipts explicitly retain `all_fixture`, nullable identity/usage and -mechanics-only limitations. The predeclared matched/held-out comparison report, -reserved-launch cancellation probe, cross-version contract-epoch decision, -accepted install/native smoke and integrated full gate are still open. R7/C1 -must remain incomplete in backlog until these gates are independently closed. - -Focused offline tests cover public four-process success, strict quorum/labels, -missing/empty roles, failed mandatory checks, distinct/native identity shape, -objective outcome uniqueness, cancellation, quota exhaustion, restart, guided -claim/submission crashes, expired exact-intent recovery and competing claims. -The final checkpoint message records the exact focused/Bash gate results; do -not infer a full-suite or native-quality pass from this partial source savepoint. - -Checkpoint gates: Council 21 tests passed in 33.055s with ResourceWarnings as -errors; router/validation/task-entry/handoff-service/MCP 67 tests passed in -20.103s (two optional SDK skips). Bash passed all 11 test files; generated -reference and `git diff --check` passed. No whole core suite was run in this -isolated slice. Preserve root's newer R5/R6 evidence when reconciling this note. - -### R7 residual source gates — October 2, 2026 - -The immutable installed R5/schema16 Service reproduced a headless Council claim -that the new Service denies: old owner acquired v28→29; the real fourth/lead -worker completed, but v38 remained awaiting that old owner. New resume correctly -refused to steal the claim. Controlled cancellation reached v44 and all owned -PIDs were absent. This proves a schema16 code guard alone is insufficient. -`017_council_contract_epoch.sql` now advances the contract epoch to17 without -new tables, activating existing exclusive-upgrade deferral and old-connection -write triggers. Historical13/16 assertions remain historical; current-version -assertions use `SUPPORTED_SCHEMA_VERSION`. Source tests prove16 active and -recoverable deferral, terminal reconciliation,17 migration, and old16 fresh/ -preopened writer rejection. Immutable installed16 versus17 confirmation and -installer/package reconciliation remain root-owned acceptance gates. - -Owned gated-launch cancellation is now tested: the actual same-user launcher -is live, cancellation remains nonterminal with no receipt, gate cleanup is -confirmed, then the cancelled Council receipt/final outcome is published. The -unstarted-runner recovery path retains Council reporting without changing the -ordinary M2 ownership boundary or declaring cancellation before cleanup. - -The separate predeclaration/comparison module now executes actual public -branch-review control (one worker) and host Council (three workers) on distinct -matched and held-out questions. It freezes task/profile/policy/prompt versions, -checks original inputs and exact terminal receipt hashes, rejects reused runs -or relabelled identical contracts as held-out, and preserves terminal ledger -immutability. Report conclusion is **inconclusive**, automatic OFF. Accepted -quality, escaped defects, rework, native quota and host usage are unmeasured; -fixture host acceptance is not native quality. Portable declaration/report: -`evidence/r7-public-fixture-workflow-{predeclaration,comparison}.json`. -Canonical private report SHA256: -`ac1deb4ea74ea70b0c337d43d0b0905bb8ed2432445e22cd3c71fb24b5fe8565`. -Exact private receipts/ledger remain at -`/Users/Dikshant/.devsquad/private-probes/devsquad-council-kpfg2i2w`. - -Native Council now fails capability-unavailable **before role launch** while -exact sandbox HTTPS attestation is unverified. Doctor shows implemented partial -mechanics, native-ready false. The last nongenerating probe and demonstrated -versus hypothetical denied facilities are recorded in -`evidence/R7-NATIVE-BOUNDARY-BLOCKER.md`; initialization and fallback/cached -catalog entries cannot lift this gate. Root owns subsequent bounded backend -attestation and MCP inventory/native/install proof; no generating retry or -broad filesystem/network relaxation is authorized by these source results. - -Residual checkpoint gates:25 Council tests passed in90.724s with ResourceWarnings -as errors;35 migration/capacity/decision/lifecycle/learning/assignment tests passed -in54.469s;12 handoff-store tests passed in4.528s;10 supervisor/reservation-crash -tests passed in1.819s. A separate retained public comparison run passed all -assertions and saved the portable report above. Bash all11 files, generated -reference and diff checks passed. Earlier comparison attempts exposed a frozen -package-path mismatch and an attempted write to a terminal run; repaired by -checking the actual packaged module and saving a separate immutable comparison -artifact, not weakening terminal fences. No whole suite/native-quality acceptance -was claimed. Root must reconcile R6 atomic finish markers/human formatter and -complete the install/native/full gates before closing R7/C1. - -### R7 actual temporary installation gate - -The focused installer regression uses `git archive bf3de086:plugin/core`, whose -full source digest exactly matches accepted immutable R5 -`90f1e87fb9acf7756dad46780672de9b84fe75e3631322857a310f089280345e`; -the actual temporary installed old package re-observed code-package digest -`e03bf3a2e362b3aff6d0a5dd00bd837f15afb216d4c58ac1d9c776c4d8f09c67` -and supported/current schema16. No lower-version rewriting was used in this -gate. Temporary MCP installation was intentionally omitted: core/Service/ -database fencing is exercised without network/dependency installation. - -Actual old active and recoverable runs independently deferred the17 candidate -while current selector, launcher, install-state and schema16 stayed unchanged. -Old active work was cancelled with cleanup; recoverable work resumed and -succeeded under the old release. Candidate72ae413 then activated without -premature schema migration; its first saved-result read migrated safely to17. -An old16 Store opened before migration failed its actual Council handoff claim -at the database trigger; a fresh old16 Service failed `SchemaVersionError`. -Council version/claim remained unchanged. All owned groups were absent, all -runs terminal, and the temporary install was removed. Portable observed facts: -`evidence/r7-temporary-install-epoch-proof.json`. - -The strict installer regression passed in6.698s, then a separate full assertion -probe verified the strengthened explicit runtime scope and owned-group cleanup. -The controlled in-memory lower-version Store test remains labelled mechanics -only; the exact old payload gate above supplies actual installer provenance. -An independent readiness agent also froze72ae413 and demonstrated immutable -accepted old16 Service rejection at a real headless Council critic barrier, -identical full-table digest before/after the rejection, successful real fourth -fixture lead after restoring continuation, and all four PIDs absent. Root owns -recording that independent review alongside final integrated R6/native gates. -No production/root source/ledger, global AI settings, network or native generation -was changed. Earlier installer test failure was a test-only wrong status JSON -field (`claim` instead of `claimed_by`); corrected with exact SQL claim checks. -Final focused installed-epoch plus controlled-epoch rerun:2 tests passed -in7.976s with ResourceWarnings as errors. Bash all11 test files, generated -reference and diff checks passed before checkpointing this portable regression. +The user authorized a public **GitHub release in joshidikshant/devsquad**. +Release v0.11.0 ships plugin0.11.0 and standalone core0.1.0; no registry or +hosted service. PR1: https://github.com/joshidikshant/devsquad/pull/1. +Publication is pending final public CI, merge and exact-commit release assets. +The whole engineering-team plan remains incomplete. + +## Accepted gates — do not repeat unchanged + +- Frozen core b83dda7fbf691503d3adf3c9ea6ecebd4071ba16 passed604 tests/ + 793.462s, two optional SDK skips, zero failures/errors/unraisable diagnostics. + Core tree62eea7fa31153ef732fd5e9dfd97951d95b367d8 is unchanged by the + subsequent legacy Bash/CI repair and documentation commits. +- Actual pristine accepted-R5 trusted-check preflight passed60 tests/33.182s + with unchanged tracked/all-input fingerprints and zero undeclared bytecode. + Input fingerprints are not branch-review candidate identities. +- Exact native EOF/fixture audit cb68f3ae-bea0-4744-90c4-96f25418fb8c: + verified Codex0.159.2/gpt-6.1-sol/low/read_only, clean, succeeded22/host accept. + Four required checks and integrity passed;60 affected tests/36.416s. + Native95970 tokens are not an invoice or Plus-window forecast. +- R3/R4/R5 accepted closure evidence and earlier live Claude CLI handoff, + Claude implementation → different verified Codex review → tests, and + Grok MCP operation remain valid for their recorded candidates/scopes. +- Final legacy watchdog checkpoint3fdc6bd065c577a9ed59566573fdce399e9b585a + is integrated. Real-deadline FIFO timer, buffered cancellation and restored + caller EXIT trap preserve Bash3.2/optional-jq and original timing limits. + Root integrated259 Bash assertions/45 affected Python tests0.927s pass; + agent gates also pass; independent final + review clean with50 legacy assertions. Exact reviewed adapter/test blobs: + 1847357c19677f276b06886386e2ac93e586b203 / + 8cc3fe9f9ee9df77b2fcb2d75c76c236a1f8d13a. + Initial CI deadline drift, rejected timer designs, caller-path cleanup defect + and corrected scheduling-only fixture failure remain honestly recorded. + +## Installed runtime and surfaces + +Current immutable release: +0.1.0-py31214-303a0e0a6c87-mcp-a26bc88afbef; Python3.12.14/MCP2.2.0/schema17. +Stable launcher /Users/Dikshant/.local/bin/squad. Zero nonterminal runs before/ +after update; pre16 online backup0600/integrityOK and old releases retained. +Reinstallchangedfalse/all driftfalse/manifests match/pipcheck pass, zero downloads. +Nine actual installed SDK tests/3.126s pass with zero skips/failures/errors. +Doctor ready for branch review and Claude→Codex delivery; four registrations +match, subscription authentication verified for Claude/Codex. + +Updated Antigravity means **agy CLI1.2.14**, NOT IDE. Actual +Gemini3.8FlashLow devsquad/squad_status observed cb68 succeeded22 in18.654s. +It read only its own MCP schema/generated result, not project files/other +servers; no zero-file-read or automatic-worker claim. +Grok/agy automatic worker readiness remains unknown; MCP proof is separate. + +This chat's existing Codex MCP server returned SCHEMA_UNSUPPORTED because it +still uses an old package. New CLI and agy succeed on the same saved run. +A scoped DevSquad-only reconnect request is pending; do not change global +settings or repeatedly probe the unchanged old connection. This does not block +the public CLI release. Local Claude plugin is still0.10.0: after main is +released run its normal scoped plugin updater and verify0.11.0. + +## Exact next action + +1. Root integrated Bash259/affected45/reference/diff/JSON gates pass; + checkpoint using git-safety. Core/full/native/installed gates need no repeat. +2. Push codex/engineering-team, require final exact-head public CI green: + legacy Bash3.2, core Python3.11 and3.14, optional MCP. Earlier CI + 37082086201/37082130929 was cancelled after true legacy failure; optional + MCP passed, cancelled core is NOT accepted. Core now fetches full history + for immutable migration fixtures and uses the tracked failfast runner. +3. Merge PR1 with exact-head fence without deleting engineering branch. Build + source/plugin archives from exact merged commit; old b83 archives are stale. + The unchanged-core wheel may be reused only after tree/hash verification. +4. Publish v0.11.0 on that exact merge SHA with verified assets/SHA256SUMS. + Verify tag, published-not-draft state and downloaded asset hashes. Update + release evidence/backlog/this checkpoint and local Claude plugin. End clean. + +## Residual scope and boundaries + +Council mechanics/schema17/reconciliation are integrated and offline tested. +Final nongenerating diagnostic is exhausted: sandbox network_request_failed/ +noHTTP, unsandboxed control strict-response rejection; zero generating calls. +Native Council remains unavailable; fixture comparison inconclusive, automatic +OFF. No additional network attempts or permission widening. +Jev OFF; the single authorized pilot is spent. Laya is conditional, not adopted. +Claude desktop Code-tab proof remains unverified/TCC; CLI is not a substitute. +No credit purchase, reset redemption, paid API fallback or global AI settings +change. Raw provider diagnostics/credentials remain outside tracked evidence. + +Evidence: evidence/public-release-0.11.0.json, +evidence/legacy-watchdog-release-repair.json, R6-terminal-readiness-partial, +R7 reconciliation/final-network diagnostics, R8-installed-workflows and +R3/R4/R5 closure files. Historical recovery detail is preserved in Git. diff --git a/docs/plans/engineering-team/backlog.json b/docs/plans/engineering-team/backlog.json index eb5cb6c..7cc1106 100644 --- a/docs/plans/engineering-team/backlog.json +++ b/docs/plans/engineering-team/backlog.json @@ -13,16 +13,16 @@ "review_followup": "SOL-REVIEW-FOLLOWUP.md", "requested_delivery_scope": ["M1", "M2", "M3", "M4", "M5", "M6", "M7", "C1"], "status": "in_progress", - "next_milestone": "M5", - "next_work_package": "R6", + "next_milestone": "M7", + "next_work_package": "R8", "partial_implementation_checkpoint": { "recorded_on": "2026-10-02", - "baseline_revision": "dfe9976", + "baseline_revision": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", "status": "partial", - "artifact": "evidence/R5-closure-2026-10-02.json", - "scope": "R3/R4/R5 accepted; installed schema16 objective outcomes/public trials with 504-test full gate, exact native repair review, safe old-schema upgrade tests and nine installed SDK tests. Earlier accepted Claude delivery/handoff and Grok/Gemini receipts retained separately.", - "next_action": "Integrate and verify delegated R6 terminal/readiness changes, then R7/C1 Council and final R8 acceptance. Do not repeat unchanged accepted proofs.", - "limitations": "Whole-plan closure is not claimed. Claude Code-tab proof remains unverified with latest TCC control denial. User clarified Antigravity CLI, not IDE; final updated agy CLI proof remains, not IDE trust. Grok/Gemini MCP status does not prove automatic writer/reviewer roles. Jev remains off with its one-request allowance spent; no paid API fallback or reset use authorized." + "artifact": "evidence/public-release-0.11.0.json", + "scope": "R3-R6 core runtime accepted: frozen604-test core, exact native EOF audit clean/accepted, pristine trusted-check integrity, installed schema17/no drift, nine actual installed SDK tests, updated agy1.2.14 actual MCP status. Legacy watchdog3fdc6bd independently reviewed and integrated without core source changes. Earlier accepted Claude delivery/handoff and Grok receipts retained separately.", + "next_action": "Complete final integrated Bash/affected checks, exact-head public CI, PR1 merge and verified v0.11.0 GitHub release; normal local Claude plugin update. Preserve original R7/C1 and R8 residual gates separately; do not repeat unchanged accepted proofs.", + "limitations": "Whole-plan closure is not claimed. Council is integrated/offline-tested but native-unavailable, comparison inconclusive and automaticOFF. Claude Code-tab proof remains unverified/TCC. Antigravity means agy CLI, not IDE; Grok/agy MCP status does not prove automatic writer/reviewer roles. Current CodexApp MCP connection needs DevSquad-only reconnect after schema17 upgrade. Jev off/single pilot spent; no paid API fallback or reset use authorized." }, "planning_checkpoint": { "recorded_on": "2026-10-01", @@ -47,9 +47,9 @@ {"id": "R3", "title": "Independent experiment and held-out evidence", "status": "complete", "milestones": ["M6"], "depends_on": [], "items": ["F3"], "evidence": "evidence/R3-closure-2026-10-02.json", "checkpoint": "Accepted immutable 39b95f1 R3 package: independent verified native Codex review clean; mandatory diff/Bash/51 affected tests pass, unchanged integrity. Six R3 source blobs exactly match prior accepted 477-test full candidate. Correction/race/revision, stale rollback/fallback and historical public read/proposal proof matrix complete. R5 public controller/outcome integration and R8 desktop acceptance remain separate."}, {"id": "R4", "title": "Normal routing, catalog lifecycle and quota", "status": "complete", "milestones": ["M6", "M7"], "depends_on": ["R1", "R2", "R3"], "items": ["F4", "G2"], "evidence": "evidence/R4-closure-2026-10-02.json", "checkpoint": "Accepted source 913abc1 and installed scoped native catalog/quota package. High shared-pool partition finding repaired and independently re-reviewed clean. 102 focused, 484 full tests (two optional SDK skips/no unraisable), 227 Bash assertions, installed normal dry-run and nine installed SDK transport tests pass. Account-wide capacity fence remains separate from discovery/qualification scopes. R5/R6/R7/R8 remain separate."}, {"id": "R5", "title": "Public saved-run outcomes and bounded experiments", "status": "complete", "milestones": ["M6"], "depends_on": ["R3", "R4"], "items": ["G1"], "evidence": "evidence/R5-closure-2026-10-02.json", "checkpoint": "Accepted dfe9976/source equivalent f87060b and installed schema16. Initial native findings repaired; exact follow-up e046a174 clean/accepted. Full 504 tests/511.241s, two optional SDK skips/no errors/failures/unraisable; nine actual installed SDK tests pass/no skips. Public complete trial/evaluate/qualify/promote/new-run/held-out regression/rollback, shared call/active deadline and all terminal projection origins pass. Safe old-schema upgrade and idempotence/drift gates pass. Failed history retained; automatic experimentation and Jev stay off."}, - {"id":"R6","title":"Normal terminal experience and readiness","status":"in_progress","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/R6-terminal-readiness-partial-2026-10-02.json","checkpoint":"Ownership093 and EOF4f integrated with Council reconciliation0e/schema17. Rejected native44 and d235 findings and invalidated bytecode checks retained; strict40 tests each3.12/3.14, root34diagnostics/11.159s and8reconciliation/SDK21.651s pass. Plain-wheel all-schema regression red1/2.657s then green1/2.831s. Pristine exact trusted-check proof, narrow EOF native acceptance, final combined full and safe installed acceptance pending. User authorizes public GitHub core0.11 release in existing repository; Council disabled/native-unavailable."}, - {"id": "R7", "title": "Complete Council within existing runner", "status": "in_progress", "milestones": ["C1"], "depends_on": ["R1", "R2", "R3", "R4", "R5", "R6"], "items": ["G5"], "checkpoint": "Isolated42979ba4/72ae4130/eee6db7: Council/epoch17/cancellation/comparison and actual immutable16-to17 temporary install proof pass. Active/recoverable16 defer unchanged, cancel/reconcile safely permits17; actual long-lived/fresh old16 clients reject Council mutation, ledger unchanged, new lead succeeds. Matched/heldout fixture comparison inconclusive; automatic use off, native quality/usage unknown. Native HTTPS attestation fails closed before generation; diagnostic paused for R6 ownership P1. Shared R6 authority/formatter reconciliation isolated on codex/council-integration; integrated full/native/installed acceptance remain open."}, - {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Requested runtime slice passes: safe update, actual Claude handoff, accepted Claude-to-Codex 477-test workflow, Grok MCP and final Gemini CLI/MCP. Broader dependencies, desktop UI and full closure audit remain open."} + {"id":"R6","title":"Normal terminal experience and readiness","status":"complete","milestones":["M5","M7"],"depends_on":["R1","R2","R4","R5"],"items":["G3","G4"],"evidence":"evidence/public-release-0.11.0.json","checkpoint":"Accepted core terminal/readiness package: exact native EOF cb68 succeeded22/clean, pristine60 trusted-check integrity, frozen604 full tests/793.462s and installed schema17/nine SDK tests/no drift. Independently reviewed legacy watchdog3fdc6bd integrated with unchanged core tree;259 Bash/45 affected agent gates pass, root integrated gates recorded in release evidence. Public CI/publication remain release gates; desktop UI/Council residuals remain separate."}, + {"id":"R7","title":"Complete Council within existing runner","status":"in_progress","milestones":["C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":["G5"],"checkpoint":"Council reconciliation0e7dd319/EOF4f integrated, schema17 actually installed; mechanical comparison/cancel/recovery/old16 fences and frozen604 core gate pass. Final nongenerating diagnostic exhausted: sandbox network_request_failed/noHTTP and unsandboxed strict-response rejection, zero generating calls. Native quality/usage unavailable; fixture comparison inconclusive; automaticOFF. No more network attempts or permission widening. Original C1 native acceptance remains incomplete, not a core-publication blocker under user-selected release scope."}, + {"id":"R8","title":"Installed proofs, external gates and closure audit","status":"in_progress","milestones":["M4","M5","M6","M7","C1"],"depends_on":["R1","R2","R3","R4","R5","R6"],"items":[],"note":"Core installed proofs may proceed before R7; full-delivery closure also requires R7. Auth/key-dependent subgates remain separately blocked.","evidence":"evidence/R8-installed-workflows-2026-10-01.json","checkpoint":"Safe updated schema17 install/no drift/nine actual SDK tests and updated agy1.2.14 actual saved-run MCP proof accepted; earlier live Claude handoff/delivery and Grok MCP retained. PR1/v0.11.0 public release pending exact-head CI/merge/assets. Existing CodexApp old MCP server needs scoped reconnect; Claude desktop Code-tab/TCC and whole-plan closure remain separate."} ], "milestones": [ { @@ -443,7 +443,7 @@ "availability": "tracked_fixture_and_tests" } ], - "blocker": "Independent R3-R5 evidence, routing, learning and observation integration remains. The single-request Jev smoke is complete but does not prove production quality; runtime guidance remains off and Laya remains conditional." + "blocker": "R3-R5 evidence/routing/learning/outcome integration is accepted and installed. Whole-milestone closure audit remains separate. Jev single-request smoke does not prove production quality; runtime off and Laya conditional." }, { "id": "M7", @@ -451,7 +451,7 @@ "depends_on": ["M6"], "status": "in_progress", "acceptance_section": "M7 — Package, migrate and prove every requested surface", - "open_review_items": ["F4", "G3", "G4"], + "open_review_items": [], "evidence": [ { "kind": "packaging_checkpoint", @@ -490,7 +490,7 @@ "availability": "portable_redacted" } ], - "blocker": "R4 catalog/quota, remaining R6 UX and desktop UI acceptance remain open. Actual Claude delivery/handoff, Grok MCP and Gemini CLI/MCP receipts pass; old login/expired-Grok blockers are superseded." + "blocker": "Core R4/R6 and updated schema17 installation are accepted. Actual Claude CLI delivery/handoff, Grok MCP and updated agy1.2.14 MCP receipts pass. Claude desktop Code-tab remains unverified/TCC; current CodexApp DevSquad-only MCP reconnect and final public CI/publication remain. No automatic Grok/agy worker claim." } ], "extensions": [ @@ -499,11 +499,11 @@ "title": "Selective evidence-based Council decisions", "optional": true, "depends_on": ["M6"], - "status": "pending", + "status": "in_progress", "specification": "SELECTION-AND-COUNCIL.md", "acceptance_section": "C1 — Optional Council extension after M6", - "evidence": [], - "blocker": null + "evidence": [{"kind":"integrated_partial_acceptance","revision":"b83dda7fbf691503d3adf3c9ea6ecebd4071ba16","artifact":"evidence/r7-final-nongenerating-network-diagnostic.json","outcome":"Integrated schema17 Council mechanics/offline tests and actual install pass. Native generating acceptance unavailable; fixture comparison inconclusive, automaticOFF. No more network attempts authorized."}], + "blocker": "Native Council network/response attestation unavailable after final nongenerating diagnostic; original C1 acceptance incomplete, not a user-selected core release blocker." } ] } diff --git a/docs/plans/engineering-team/evidence/legacy-watchdog-release-repair.json b/docs/plans/engineering-team/evidence/legacy-watchdog-release-repair.json new file mode 100644 index 0000000..012903f --- /dev/null +++ b/docs/plans/engineering-team/evidence/legacy-watchdog-release-repair.json @@ -0,0 +1,111 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "Isolated portable legacy watchdog and narrow offline CI plumbing repair; not release publication or whole-plan acceptance.", + "baseline_revision": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "branch": "codex/release-watchdog", + "preserved_branch": "codex/council-integration", + "implementation_blobs": { + "plugin/lib/adapter.sh": "1847357c19677f276b06886386e2ac93e586b203", + "test/test_m1_legacy.sh": "8cc3fe9f9ee9df77b2fcb2d75c76c236a1f8d13a", + ".github/workflows/offline.yml": "2d231113b575a2568d3c51844f4551f0f30d7c70" + }, + "watchdog_diff_sha256": "51fd52c51c51b870d18d98d4d573b3031e0bd7414d9ee85f9434286c703b8cc7", + "cause": "The original portable watchdog decremented timeout_secs * 20 polls. Each external ps/tr probe and sleep consumed additional elapsed time, so configured deadlines drifted on macOS Bash 3.2.", + "repair": { + "deadline": "Background Bash builtin read -t on a private duplex FIFO has a real deadline independent of process-probe/polling cost. The parent observes timer completion before reading its fully published timeout/error marker.", + "cancellation": "A command-scoped parent FD9 writes cancel and remains duplex-open through wait. Cancellation before the timer opens remains buffered; cancellation after natural timer exit cannot block. No numeric timer signals or external sleep timer are used.", + "ownership": "Bash's owned job table observes direct-child completion. Only the existing CLI/descendant timeout cleanup receives TERM/KILL; no CLI wrapper or new process-authority contract was introduced.", + "resources": "mktemp creates the private mode700 directory; FIFO mode is600. Timer stdio is /dev/null, its inherited EXIT trap is reset, and the private FIFO is not inherited by the CLI or descendants. Command-local cancellation redirection restores caller FD9.", + "exit_scope": "The original EXIT trap is captured with builtin trap -p. Active-scope interruption still cancels/reaps the timer and cleans owned paths. Normal completion reads and removes capture files, then restores the shell-generated original trap before any classification return; caller globals cannot become cleanup targets.", + "compatibility": "Stock macOS Bash3.2, optional jq, no GNU timeout or Python runtime dependency added. Actual CLI nonzero status remains visible in CLI_ERROR; actual deadline retains TIMEOUT and normalized wrapper exit1. Existing descendant escalation and timing thresholds are unchanged." + }, + "failure_history": [ + { + "kind": "reported_CI_release_blocker", + "reported_by": "root", + "run_id": "37082130929", + "job_id": "111084739785", + "observation": "macOS legacy gate reported 4s for a1s deadline and6.123s/8.644s for a3s TERM-ignoring root; existing bounds were <4s/<5s.", + "agent_log_fetch": "gh run view --log-failed reported logs unavailable while the job was still running; no direct CI pass or complete-log inspection is claimed." + }, + { + "kind": "controlled_original_deadline_red", + "probe_delay_seconds": 0.2, + "deadline_seconds": 1, + "observed_elapsed_seconds": 5.848, + "initial_legacy_gate": "21 passed/2 failed: delayed-probe deadline plus an initial timer-observation assertion incompatible with the old poll-only implementation. The latter is instrumentation history, not another product defect." + }, + { + "kind": "rejected_uncommitted_timer_designs", + "observation": "TERM+wait on a sleep timer inherited ignored TERM and took2.096s on the fast path (independent helper2.025s). A jobs snapshot followed by numeric timer kill still had a stale-PID race; the independent signal observer recorded a former PID with kernel_live=false without sending a real unowned signal. These designs were replaced, not accepted." + }, + { + "kind": "controlled_EXIT_scope_red", + "prior_adapter_blob": "0789aae95e52272c5eb37e65745fdbb114486342", + "legacy_gate": "44 passed/6 failed", + "observation": "After invocation locals unwound, the new EXIT trap deleted same-named caller timer paths and lost the prior caller EXIT hook on portable success, GNU success and portable natural failure. All six regressions now pass." + }, + { + "kind": "test_only_scheduling_failure", + "prior_test_blob": "e6534aebed4fd16fd54b738b95e4c8e862650ab7", + "bash_gate": "1/11 files failed; legacy49 passed/1 failed", + "observation": "A fixed .2s fixture delay did not always force cancellation before FIFO open under combined-gate scheduling. Product timing bounds passed. The fixture now positively gates timer entry until the actual cancellation write succeeds, with parent FD9 retained; no timing assertion was widened." + } + ], + "verified_gates": { + "legacy_focused": { + "command": "/bin/bash test/test_m1_legacy.sh", + "bash": "3.2.57(1)-release arm64-apple-darwin26", + "assertions": 50, + "failures": 0, + "coverage": ["fast completion <1s", "inherited ignored TERM <1s", "actual buffered cancellation before timer FIFO open", "already-exited/reaped timer cancellation without numeric signals", "private directory/FIFO permissions and cleanup", "caller FD9 preservation", "provider/descendant without private FIFO FD", "CLI exit7 retained in CLI_ERROR", "caller EXIT hook and sentinel paths preserved on both branches", "active-scope exit17 cancels/reaps timer", "TERM-resistant root and descendants removed", "deliberately slow ps does not extend1s deadline"] + }, + "bash_all": { + "command": "/bin/bash test/run.sh", + "files": 11, + "assertions": 259, + "failures": 0, + "timings_seconds": {"fast": 0.273, "ignored_TERM": 0.197, "buffered_pre_open_cancel": 0.209, "TERM_resistant_3s": 3.246, "delayed_ps_1s": 1.243} + }, + "offline_python_affected": { + "command": "DEVSQUAD_BUILD_PYTHON=/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3 PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core /Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3 -m unittest test_m1 test_m1_gate_review test_adapters -q", + "tests": 45, + "seconds": 0.711, + "failures": 0, + "errors": 0, + "skips": 0 + }, + "other": ["/bin/bash -n plugin/lib/adapter.sh test/test_m1_legacy.sh", "scripts/generate-core-reference.py --check", "git diff --check"] + }, + "independent_review": { + "reviewer": "r6_readiness", + "scope": "Frozen adapter and legacy test only", + "result": "clean; all scoped findings resolved", + "legacy_assertions": 50, + "failures": 0, + "timings_seconds": {"fast": 0.194, "ignored_TERM": 0.202, "buffered_pre_open_cancel": 0.216, "TERM_resistant_3s": 3.236, "delayed_ps_1s": 1.244}, + "before_after_hashes_unchanged": true, + "confirmed": "The original private caller-path deletion reproduction now preserves its markers after shell EXIT. The final test-only gate releases FIFO open strictly after a successful cancellation write." + }, + "CI_plumbing": { + "core_checkout": "actions/checkout@v4 with fetch-depth:0 supplies historical archived installer fixtures.", + "core_runner": "python scripts/run-core-tests.py --failfast retains complete discovery, spawn-safe main, unraisable diagnostics and fail-fast behavior.", + "controlled_archive_proof": { + "clone": "Unique private local file:// --depth1 --single-branch clone of codex/release-watchdog; exact HEAD b83dda7, is-shallow=true; no network.", + "archives": ["f4fa657:plugin/core", "bf3de0867484552d354e6e8b6ba835f31a93aeb3:plugin/core"], + "shallow_exit_codes": [128, 128], + "shallow_error": "fatal: not a valid object name", + "original_full_history_exit_codes": [0, 0] + }, + "remote_CI_rerun": "Pending root integration/publication; no remote CI pass claimed." + }, + "unchanged_core": { + "plugin_core_tree": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "python_tests_tree": "669eaf8487a7f2f983305162da33c1bda53116b5", + "core_runner_blob": "011416767e4e145fce6a49a0a38444cad49a68b7", + "prior_root_full_gate": "Root reported604 tests/793.462s, two optional SDK skips, zero failures/errors/unraisable on frozen b83dda7. This agent did not repeat the full suite." + }, + "boundaries": "No root source/ledger/installation changes, live providers, native audits, full core suite, publication or global AI settings. RESUME/backlog remain root-owned. Private controlled fixtures are outside Git.", + "next_action": "Root reviews CI config and integrates this checkpoint, records authoritative RESUME, rebuilds only changed final source/plugin assets, and owns remaining release/CI gates." +} diff --git a/docs/plans/engineering-team/evidence/public-release-0.11.0.json b/docs/plans/engineering-team/evidence/public-release-0.11.0.json index a56dbc0..6a7a1aa 100644 --- a/docs/plans/engineering-team/evidence/public-release-0.11.0.json +++ b/docs/plans/engineering-team/evidence/public-release-0.11.0.json @@ -1,60 +1,192 @@ { - "schema_version":1, - "recorded_on":"2026-10-02", - "status":"release_gates_in_progress_not_published", - "authorization":"User chose public GitHub release in existing joshidikshant/devsquad; no package registry or hosted service", - "pull_request":"https://github.com/joshidikshant/devsquad/pull/1", - "frozen_core_candidate":"b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", - "frozen_core_tree":"62eea7fa31153ef732fd5e9dfd97951d95b367d8", - "native_eof_followup":{ - "run_id":"cb68f3ae-bea0-4744-90c4-96f25418fb8c", - "base":"8242eff68b24538aacc1b40b67b2e88e96b04d52", - "target":"b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", - "candidate_sha256":"f3243afeae0f011321d69fca1dc2f2225e4651f2a35e3452c1266cfa3da58791", - "verdict":"clean", "findings":[],"state":"succeeded","version":22,"host_disposition":"accept", - "identity":{"harness":"codex","harness_version":"codex-cli0.159.2","model_id":"gpt-6.1-sol","effort":"low","verification":"verified","permission_policy":"read_only"}, - "checks":{"diff":"passed","bash":"passed","affected_tests":60,"seconds":36.416,"reference":"passed","all_required":true,"all_integrity":"verified_unchanged"}, - "artifact_hashes_verified":["0dcb6d16b34c4f9c6e3eed6aa7e0161bf1d46c5f7f61370a7f8b80008dae5b69","46b3f1b3c8c1d8a7e824e4fef4502eb88841f27e1deaddf1b166359937f97283","d0f0b5b3e2169e34ed4ae63eacc7cdedce82b68fd9de4a904d01323f6910253f","182acd24afa59bc37c539f6fa8f0e6b1dd01275cfcd784970c2ffd8e21f00e32"], - "usage":{"source":"native_reported","input_tokens":95453,"output_tokens":517,"total_tokens":95970}, - "limits":"Narrow EOF/fixture audit only, not whole-plan or native Council acceptance; usage is not invoice/Plus-window estimate" - }, - "first_ci":{ - "push_run":37082086201,"pr_run":37082130929, - "status":"cancelled_superseded_after_confirmed_legacy_failure", - "legacy":"Actual Bash3.2 deadline failure: poll-count timing stretched1s to4s and3s to6.123s/8.644s", - "optional_mcp":"passed_both_runs", - "core":"cancelled_not_accepted; shallow checkout cannot provide immutable f4fa657/bf3 migration fixtures", - "repair":"Isolated actual sleep deadline and unchanged timing bounds; full-history core checkout and tracked spawn-safe failfast runner; independent review pending", - "no_passing_failure_claim":true - }, - "fresh_install_b83":{ - "actual_core_only":true,"version":"squad0.1.0","changed":true,"reinstall_changed":false,"status_changed":false,"all_drift":false,"manifest_matches":true, - "digest":"303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52", - "wheel_schemas":11,"wheel_migrations":17,"all_packaged_bytes_match_source":true, - "proof_sha256":"a0c7914ad5228276834ac242602fd510601f65562e406dd3dd75b31d107931ed", - "initial_failed_gate":"Whole-production filesystem equality invalidated by separately authorized concurrent root native ledger activity; perfield attribution unavailable, no unchanged-ledger claim", - "subsequent_validation":"Stable selector/manifest/installstate/launcher observed unchanged; unique private install/runtime/bin only", - "provider_or_network_calls":0,"production_update":false - }, - "final_full":{"target":"b83dda7fbf691503d3adf3c9ea6ecebd4071ba16","exec_session":12430,"tests":604,"seconds":793.462,"monotonic_seconds":793.8633149590023,"utc_seconds":793.881373,"failures":0,"errors":0,"optional_sdk_skips":2,"unraisable":[],"status":"passed","source_tests_scripts_frozen":true}, - "installation":{ - "release":"0.1.0-py31214-303a0e0a6c87-mcp-a26bc88afbef","source_digest":"303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52", - "python":"3.12.14","mcp":"2.2.0","ledger_schema":17,"nonterminal_runs_before_and_after":0,"private_pre16_online_backup":"retained0600_integrity_ok", - "first_changed":true,"reinstall_changed":false,"all_drift":false,"manifest_matches":true,"pip_check":"passed","previous_releases_retained":true,"downloads":0, - "installed_sdk":{"tests":9,"seconds":3.126,"failures":0,"errors":0,"skips":0,"origin":"selected immutable release/core/src, not primary checkout"}, - "sdk_harness_first_assertion":"Initial helper assumed site-packages origin and aborted before tests; installer intentionally imports its immutable copied core/src. Corrected origin assertion retains release-bound/no-primary-source check; nine tests then passed.", - "doctor":{"ready":true,"branch_review_ready":true,"issue_delivery_ready":true,"matching_registrations":4,"claude_subscription_authenticated":true,"codex_subscription_authenticated":true,"council_native_ready":false,"grok_antigravity_worker_readiness":"unverified/unknown, not implied by MCP registration"}, - "saved_run_continuity":{"run_id":"cb68f3ae-bea0-4744-90c4-96f25418fb8c","state":"succeeded","version":22} - }, - "updated_antigravity_cli":{ - "version":"1.2.14","observed_init_model":"gemini-3.8-flash-low","mode":"plan/sandbox; native request-review","operation":"one devsquad/squad_status call","run_id":"cb68f3ae-bea0-4744-90c4-96f25418fb8c","observed_state":"succeeded","observed_version":22, - "seconds":18.65435924999838,"returncode":0,"timed_out":false,"native_result":"SUCCESS", - "stdout_sha256":"cc6045878de2124b845adab33401ecd1f52317dd1f6e654b2b439041b2da1863","stderr_sha256":"3a7f8550106d86abb322e8ed0fb30a0296c0e97c6be03d5a8033b99d16737459", - "actual_mcp_output_sha256":"d22dccedbcf001999c59d6167820e04a833484f70a2d738e2e7f5c708b9a98f8", - "native_file_reads":"Own DevSquad MCP schema and provider-generated tool result file only; not zero file reads. Actual tool-result envelope independently parsed and matched saved run; no project files, other servers or mutations.", - "limits":"CLI MCP operation proof only; not Antigravity IDE, automatic writer/reviewer, cost savings or whole-plan acceptance" - }, - "codex_app_existing_connection":{"operation":"squad_status cb68","result":"SCHEMA_UNSUPPORTED","reason":"existing long-running server is bound to old package; reconnect only DevSquad connection to select new release","new_cli_and_agy_same_run":"succeeded22","user_action":"async reconnect request pending; no global restart/settings changes"}, - "limitations":{"council":"native-unavailable, automaticOFF, fixture comparison inconclusive; no more network attempts","jev":"OFF; authorized single pilot spent","claude_desktop_code_tab":"unverified/TCC, not substituted by CLI proof","antigravity":"agyCLI1.2.14 updated-release MCP status verified; automatic worker roles unverified and IDE out of scope","whole_plan":"not complete"}, - "remaining":"Integrate independently reviewed legacy deadline/CI repair; final green public CI; merge/tag/release verified assets. Existing CodexApp connection needs scoped reconnect; core604/native/installed9SDK/agyCLI gates already passed, do not repeat unchanged." + "schema_version": 1, + "recorded_on": "2026-10-02", + "status": "integrated_release_candidate_public_ci_pending", + "authorization": "User chose public GitHub release in existing joshidikshant/devsquad; no package registry or hosted service", + "pull_request": "https://github.com/joshidikshant/devsquad/pull/1", + "frozen_core_candidate": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "frozen_core_tree": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "native_eof_followup": { + "run_id": "cb68f3ae-bea0-4744-90c4-96f25418fb8c", + "base": "8242eff68b24538aacc1b40b67b2e88e96b04d52", + "target": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "candidate_sha256": "f3243afeae0f011321d69fca1dc2f2225e4651f2a35e3452c1266cfa3da58791", + "verdict": "clean", + "findings": [], + "state": "succeeded", + "version": 22, + "host_disposition": "accept", + "identity": { + "harness": "codex", + "harness_version": "codex-cli0.159.2", + "model_id": "gpt-6.1-sol", + "effort": "low", + "verification": "verified", + "permission_policy": "read_only" + }, + "checks": { + "diff": "passed", + "bash": "passed", + "affected_tests": 60, + "seconds": 36.416, + "reference": "passed", + "all_required": true, + "all_integrity": "verified_unchanged" + }, + "artifact_hashes_verified": [ + "0dcb6d16b34c4f9c6e3eed6aa7e0161bf1d46c5f7f61370a7f8b80008dae5b69", + "46b3f1b3c8c1d8a7e824e4fef4502eb88841f27e1deaddf1b166359937f97283", + "d0f0b5b3e2169e34ed4ae63eacc7cdedce82b68fd9de4a904d01323f6910253f", + "182acd24afa59bc37c539f6fa8f0e6b1dd01275cfcd784970c2ffd8e21f00e32" + ], + "usage": { + "source": "native_reported", + "input_tokens": 95453, + "output_tokens": 517, + "total_tokens": 95970 + }, + "limits": "Narrow EOF/fixture audit only, not whole-plan or native Council acceptance; usage is not invoice/Plus-window estimate" + }, + "first_ci": { + "push_run": 37082086201, + "pr_run": 37082130929, + "status": "cancelled_superseded_after_confirmed_legacy_failure", + "legacy": "Actual Bash3.2 deadline failure: poll-count timing stretched1s to4s and3s to6.123s/8.644s", + "optional_mcp": "passed_both_runs", + "core": "cancelled_not_accepted; shallow checkout cannot provide immutable f4fa657/bf3 migration fixtures", + "repair": "Integrated3fdc6bd FIFO real-deadline watchdog, unchanged timing limits/caller EXIT restoration; full-history checkout and tracked spawn-safe failfast runner. Independent final source review clean.", + "no_passing_failure_claim": true + }, + "fresh_install_b83": { + "actual_core_only": true, + "version": "squad0.1.0", + "changed": true, + "reinstall_changed": false, + "status_changed": false, + "all_drift": false, + "manifest_matches": true, + "digest": "303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52", + "wheel_schemas": 11, + "wheel_migrations": 17, + "all_packaged_bytes_match_source": true, + "proof_sha256": "a0c7914ad5228276834ac242602fd510601f65562e406dd3dd75b31d107931ed", + "initial_failed_gate": "Whole-production filesystem equality invalidated by separately authorized concurrent root native ledger activity; perfield attribution unavailable, no unchanged-ledger claim", + "subsequent_validation": "Stable selector/manifest/installstate/launcher observed unchanged; unique private install/runtime/bin only", + "provider_or_network_calls": 0, + "production_update": false + }, + "final_full": { + "target": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", + "exec_session": 12430, + "tests": 604, + "seconds": 793.462, + "monotonic_seconds": 793.8633149590023, + "utc_seconds": 793.881373, + "failures": 0, + "errors": 0, + "optional_sdk_skips": 2, + "unraisable": [], + "status": "passed", + "source_tests_scripts_frozen": true + }, + "installation": { + "release": "0.1.0-py31214-303a0e0a6c87-mcp-a26bc88afbef", + "source_digest": "303a0e0a6c87ff472d1d7b52cb63ea56798a11f345f483736a18669c5f4d2a52", + "python": "3.12.14", + "mcp": "2.2.0", + "ledger_schema": 17, + "nonterminal_runs_before_and_after": 0, + "private_pre16_online_backup": "retained0600_integrity_ok", + "first_changed": true, + "reinstall_changed": false, + "all_drift": false, + "manifest_matches": true, + "pip_check": "passed", + "previous_releases_retained": true, + "downloads": 0, + "installed_sdk": { + "tests": 9, + "seconds": 3.126, + "failures": 0, + "errors": 0, + "skips": 0, + "origin": "selected immutable release/core/src, not primary checkout" + }, + "sdk_harness_first_assertion": "Initial helper assumed site-packages origin and aborted before tests; installer intentionally imports its immutable copied core/src. Corrected origin assertion retains release-bound/no-primary-source check; nine tests then passed.", + "doctor": { + "ready": true, + "branch_review_ready": true, + "issue_delivery_ready": true, + "matching_registrations": 4, + "claude_subscription_authenticated": true, + "codex_subscription_authenticated": true, + "council_native_ready": false, + "grok_antigravity_worker_readiness": "unverified/unknown, not implied by MCP registration" + }, + "saved_run_continuity": { + "run_id": "cb68f3ae-bea0-4744-90c4-96f25418fb8c", + "state": "succeeded", + "version": 22 + } + }, + "updated_antigravity_cli": { + "version": "1.2.14", + "observed_init_model": "gemini-3.8-flash-low", + "mode": "plan/sandbox; native request-review", + "operation": "one devsquad/squad_status call", + "run_id": "cb68f3ae-bea0-4744-90c4-96f25418fb8c", + "observed_state": "succeeded", + "observed_version": 22, + "seconds": 18.65435924999838, + "returncode": 0, + "timed_out": false, + "native_result": "SUCCESS", + "stdout_sha256": "cc6045878de2124b845adab33401ecd1f52317dd1f6e654b2b439041b2da1863", + "stderr_sha256": "3a7f8550106d86abb322e8ed0fb30a0296c0e97c6be03d5a8033b99d16737459", + "actual_mcp_output_sha256": "d22dccedbcf001999c59d6167820e04a833484f70a2d738e2e7f5c708b9a98f8", + "native_file_reads": "Own DevSquad MCP schema and provider-generated tool result file only; not zero file reads. Actual tool-result envelope independently parsed and matched saved run; no project files, other servers or mutations.", + "limits": "CLI MCP operation proof only; not Antigravity IDE, automatic writer/reviewer, cost savings or whole-plan acceptance" + }, + "codex_app_existing_connection": { + "operation": "squad_status cb68", + "result": "SCHEMA_UNSUPPORTED", + "reason": "existing long-running server is bound to old package; reconnect only DevSquad connection to select new release", + "new_cli_and_agy_same_run": "succeeded22", + "user_action": "async reconnect request pending; no global restart/settings changes" + }, + "limitations": { + "council": "native-unavailable, automaticOFF, fixture comparison inconclusive; no more network attempts", + "jev": "OFF; authorized single pilot spent", + "claude_desktop_code_tab": "unverified/TCC, not substituted by CLI proof", + "antigravity": "agyCLI1.2.14 updated-release MCP status verified; automatic worker roles unverified and IDE out of scope", + "whole_plan": "not complete" + }, + "remaining": "Push final candidate; require exact-head public CI green, merge PR1 and publish v0.11.0 exact-commit verified assets; local Claude plugin normal update. Existing CodexApp connection needs scoped reconnect. Core604/native/installed9SDK/agyCLI gates already passed; do not repeat unchanged. Whole-plan/Council/desktop residuals remain separate.", + "integrated_legacy_gate": { + "source_checkpoint": "3fdc6bd065c577a9ed59566573fdce399e9b585a", + "adapter_blob": "1847357c19677f276b06886386e2ac93e586b203", + "test_blob": "8cc3fe9f9ee9df77b2fcb2d75c76c236a1f8d13a", + "independent_review": "clean; committed bytes equal independently reviewed/tested candidate", + "root_bash": { + "assertions": 259, + "files": 11, + "failures": 0, + "timings_seconds": { + "fast": 0.267, + "ignored_TERM": 0.211, + "pre_open_cancel": 0.207, + "TERM_resistant": 3.259, + "slow_ps": 1.291 + } + }, + "root_affected_python": { + "tests": 45, + "seconds": 0.927, + "failures": 0, + "errors": 0 + }, + "reference": "passed", + "core_tree_unchanged": true, + "failure_history": "evidence/legacy-watchdog-release-repair.json" + } } diff --git a/plugin/lib/adapter.sh b/plugin/lib/adapter.sh index 22df821..1847357 100644 --- a/plugin/lib/adapter.sh +++ b/plugin/lib/adapter.sh @@ -55,6 +55,38 @@ _adapter_signal_snapshot() { done < "$snapshot_file" } +# Bash's own job table tracks our directly-owned children without spawning +# ps/tr for every poll. Timer cancellation below never signals a numeric PID. +_adapter_job_running() { + local pid="$1" running + running=$(jobs -pr) + case $'\n'"$running"$'\n' in + *$'\n'"$pid"$'\n'*) return 0 ;; + *) return 1 ;; + esac +} + +_adapter_stop_timer() { + local timer_pid="${1:-}" control_file="${2:-}" + [[ -n "$timer_pid" ]] || return 0 + # Duplex open cannot block if the timer has already exited. Keep it open + # through wait so an early cancellation remains buffered until the timer + # opens its reader. The command-local descriptor restores the caller's FD9. + # No numeric PID signal is ever sent to a timer that may have exited/reaped. + { + printf 'cancel\n' >&9 + wait "$timer_pid" 2>/dev/null || true + } 9<>"$control_file" +} + +_adapter_cleanup_timer() { + _adapter_stop_timer "${timer_pid:-}" "${timer_control_file:-}" + if [[ -n "${timer_dir:-}" ]]; then + rm -f "$timer_dir/control" "$timer_dir/deadline" "$timer_dir/processes" + rmdir "$timer_dir" 2>/dev/null || true + fi +} + # Resolve model: agent-specific (agent_models.) > # global (.preferences.) > "" (CLI default). # Values may be exact model names OR tiers ("tier:fast" / "tier:frontier"), @@ -144,7 +176,10 @@ _adapter_invoke() { local stderr_file stdout_file stderr_file=$(mktemp) stdout_file=$(mktemp) - trap 'rm -f "${stderr_file:-}" "${stdout_file:-}"' EXIT + local timer_pid="" timer_dir="" timer_control_file="" + local previous_exit_trap + previous_exit_trap=$(builtin trap -p EXIT) + trap '_adapter_cleanup_timer; rm -f "${stderr_file:-}" "${stdout_file:-}"' EXIT local exit_code=0 if [[ -n "$timeout_cmd" ]]; then @@ -157,21 +192,44 @@ _adapter_invoke() { # Portable watchdog: every call is bounded even on hosts with no # timeout/gtimeout binary (observed live: an unauthenticated CLI # waiting on OAuth blocks forever) + timer_dir=$(mktemp -d) + timer_control_file="$timer_dir/control" + local timer_deadline_file="$timer_dir/deadline" + mkfifo -m 600 "$timer_control_file" + # Bash read's real deadline is independent of polling/inspection cost. + # Its private FIFO is cancellation authority: no timer-PID signalling, + # orphan sleep, inherited ignored TERM, or retained capture descriptors. + ( + trap - EXIT + local timer_message="" timer_status=0 + if IFS= read -r -t "$timeout_secs" timer_message <>"$timer_control_file"; then + [[ "$timer_message" == "cancel" ]] || printf 'error\n' >"$timer_deadline_file" + else + timer_status=$? + if [[ "$timer_status" -eq 1 || "$timer_status" -gt 128 ]]; then + printf 'timeout\n' >"$timer_deadline_file" + else + printf 'error\n' >"$timer_deadline_file" + fi + fi + ) /dev/null 2>&1 & + timer_pid=$! if [[ -n "${ADAPTER_STDIN_FILE:-}" ]]; then "$cli" "${ADAPTER_ARGS[@]}" <"$ADAPTER_STDIN_FILE" >"$stdout_file" 2>"$stderr_file" & else "$cli" "${ADAPTER_ARGS[@]}" >"$stdout_file" 2>"$stderr_file" & fi local cli_pid=$! - local process_snapshot="${stdout_file}.processes" timed_out="false" - local polls_remaining=$(( timeout_secs * 20 )) state="" - # Poll the directly-owned child. This avoids a background sleep/watchdog - # retaining capture descriptors after fast completion. - while :; do - state=$(ps -p "$cli_pid" -o stat= 2>/dev/null | tr -d ' ' || true) - [[ -z "$state" || "$state" == Z* ]] && break - if [[ "$polls_remaining" -le 0 ]]; then - timed_out="true" + local process_snapshot="$timer_dir/processes" timed_out="false" monitor_failed="false" + while _adapter_job_running "$cli_pid"; do + # Observe completion, not an in-progress marker write. The completed + # timer has published either its actual deadline or a monitor failure. + if ! _adapter_job_running "$timer_pid"; then + if [[ "$(cat "$timer_deadline_file" 2>/dev/null || true)" == "timeout" ]]; then + timed_out="true" + else + monitor_failed="true" + fi _adapter_snapshot_tree "$cli_pid" > "$process_snapshot" _adapter_signal_snapshot "$process_snapshot" TERM sleep 0.1 @@ -179,8 +237,9 @@ _adapter_invoke() { break fi sleep 0.05 - polls_remaining=$(( polls_remaining - 1 )) done + _adapter_stop_timer "$timer_pid" "$timer_control_file" + timer_pid="" if wait "$cli_pid"; then exit_code=0 else @@ -189,13 +248,23 @@ _adapter_invoke() { if [[ "$timed_out" == "true" ]]; then _adapter_signal_snapshot "$process_snapshot" KILL exit_code=124 + elif [[ "$monitor_failed" == "true" ]]; then + exit_code=125 fi - rm -f "$process_snapshot" + _adapter_cleanup_timer + timer_dir="" + timer_control_file="" fi local stdout stderr_content stdout=$(cat "$stdout_file" 2>/dev/null) stderr_content=$(cat "$stderr_file" 2>/dev/null) + rm -f "$stderr_file" "$stdout_file" + # Cleanup uses invocation locals only while they are live. Restore the + # caller's shell-generated trap before any classification/return unwinds + # that scope, so same-named caller globals can never become cleanup targets. + builtin trap - EXIT + if [[ -n "$previous_exit_trap" ]]; then eval "$previous_exit_trap"; fi # CLI-specific auth signal (may appear on stdout with exit 0, e.g. grok's # sign-in banner) — checked before the success path diff --git a/test/test_m1_legacy.sh b/test/test_m1_legacy.sh index f715b89..8cc3fe9 100755 --- a/test/test_m1_legacy.sh +++ b/test/test_m1_legacy.sh @@ -13,6 +13,7 @@ cat > "$T/bin/codex" <<'EOF' #!/usr/bin/env bash case "${FAKE_MODE:-fast}" in fast) printf 'ok' ;; + error) printf 'deliberate offline failure\n' >&2; exit 7 ;; tree) sh -c 'trap "" TERM; echo $$ > "$DESC_PID_FILE"; while :; do sleep 1; done' & wait @@ -27,20 +28,86 @@ esac EOF chmod +x "$T/bin/codex" -# Record the portable watchdog's timer PID so the success path can prove it -# did not orphan the sleep process. Other sleep durations use the real binary. -cat > "$T/bin/sleep" <<'EOF' +# Make process inspection deliberately expensive for the deadline regression. +# The old watchdog counted probes rather than elapsed time, so every delayed +# ps call extended the configured timeout. Normal test calls remain unchanged. +cat > "$T/bin/ps" <<'EOF' #!/usr/bin/env bash -if [[ "$1" == "2" && -n "${WATCHDOG_SLEEP_PID_FILE:-}" ]]; then - printf '%s\n' "$$" > "$WATCHDOG_SLEEP_PID_FILE" +if [[ "${FAKE_PROBE_DELAY:-0}" != "0" ]]; then + /bin/sleep "$FAKE_PROBE_DELAY" fi -exec /bin/sleep "$@" +exec /bin/ps "$@" EOF -chmod +x "$T/bin/sleep" +chmod +x "$T/bin/ps" + +# A fake provider and its descendant must not retain the private FIFO FD. +# Brief real work also exercises cancellation while retaining the existing +# sub-second fast-call contract. +cat > "$T/bin/fast-echo" <<'EOF' +#!/usr/bin/env bash +[[ ! -p /dev/fd/9 ]] || exit 33 +/bin/bash -c '[[ ! -p /dev/fd/9 ]]' || exit 34 +/bin/sleep 0.05 +exec /bin/echo "$@" +EOF +chmod +x "$T/bin/fast-echo" + +# Observe only the adapter's cancellation seam: record the actual timer job +# and its private paths, call the real helper, then check its wait/cleanup. +# The gated timer-entry seam waits before FIFO open until cancellation is written. +cat > "$T/fast-call.sh" <<'EOF' +#!/usr/bin/env bash +[[ "${IGNORE_TERM:-0}" != "1" ]] || trap '' TERM +source "$1/plugin/lib/codex-wrapper.sh" +eval "$(declare -f _adapter_stop_timer | sed '1s/_adapter_stop_timer/_test_stop_timer/')" +_adapter_stop_timer() { + [[ -n "${1:-}" ]] || return 0 + printf '%s\n' "$1" > "$OBSERVATION_DIR/timer.pid" + printf '%s\n' "$2" > "$OBSERVATION_DIR/control.path" + if [[ "$(LC_ALL=C ls -ld "$2" | awk '{ print substr($1, 1, 10) }')" == "prw-------" ]] && + [[ "$(LC_ALL=C ls -ld "${2%/*}" | awk '{ print substr($1, 1, 10) }')" == "drwx------" ]]; then + : > "$OBSERVATION_DIR/private" + fi + _test_stop_timer "$@" +} +printf() { + builtin printf "$@" || return $? + if [[ "${DELAY_TIMER_OPEN:-0}" == "1" && "${1:-}" == 'cancel\n' && ! -f "$OBSERVATION_DIR/open-started" ]]; then + : > "$OBSERVATION_DIR/cancel-before-open" + fi +} +trap() { + builtin trap "$@" + # Gate the exact timer-entry seam before any FIFO redirection (not inside + # read(), whose redirect would already be applied). Release only after the + # real cancellation write succeeds, while the parent retains duplex FD9. + if [[ "${DELAY_TIMER_OPEN:-0}" == "1" && "${1:-}" == "-" && "${2:-}" == "EXIT" ]]; then + local test_polls=0 + : > "$OBSERVATION_DIR/delay-started" + while [[ ! -f "$OBSERVATION_DIR/cancel-before-open" && "$test_polls" -lt 200 ]]; do + /bin/sleep 0.01 + test_polls=$((test_polls + 1)) + done + [[ -f "$OBSERVATION_DIR/cancel-before-open" ]] || exit 19 + : > "$OBSERVATION_DIR/open-started" + fi +} +kill() { printf '%s\n' "$*" >> "$OBSERVATION_DIR/signals"; builtin kill "$@"; } +if [[ "${TRIGGER_POLL_EXIT:-0}" == "1" ]]; then + _adapter_job_running() { exit 17; } +fi +exec 9>"$OBSERVATION_DIR/caller-fd9" +s=$($NOW_BIN) +if invoke_codex hello 10 2; then call_rc=0; else call_rc=$?; fi +f=$($NOW_BIN) +awk -v s="$s" -v f="$f" 'BEGIN { printf "%.3f", f-s }' > "$OBSERVATION_DIR/elapsed" +printf 'restored\n' >&9 +jobs -pr > "$OBSERVATION_DIR/remaining-jobs" +exit "$call_rc" +EOF +chmod +x "$T/fast-call.sh" # Force the portable path even on systems with timeout/gtimeout installed. -WATCHDOG_SLEEP_PID_FILE="$T/watchdog-sleep.pid"; export WATCHDOG_SLEEP_PID_FILE -FAST_ELAPSED_FILE="$T/fast-elapsed"; export FAST_ELAPSED_FILE NOW_BIN="$T/bin/now"; export NOW_BIN cat > "$NOW_BIN" <<'EOF' #!/usr/bin/perl @@ -48,16 +115,126 @@ use Time::HiRes qw(time); printf "%.6f", time; EOF chmod +x "$NOW_BIN" -HOME="$T/home" PATH="/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="/bin/echo" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=fast \ - bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; s=$($NOW_BIN); invoke_codex hello 10 2; f=$($NOW_BIN); awk -v s="$s" -v f="$f" '\''BEGIN { printf "%.3f", f-s }'\'' > "$FAST_ELAPSED_FILE"' _ "$ROOT" > "$T/out" 2> "$T/err" -fast_elapsed=$(cat "$FAST_ELAPSED_FILE") +mkdir "$T/fast-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/fast-observed" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/fast-echo" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=fast \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/out" 2> "$T/err" +fast_elapsed=$(cat "$T/fast-observed/elapsed") grep -q 'exec hello' "$T/out" && ok || bad "portable watchdog output" awk -v e="$fast_elapsed" 'BEGIN { exit !(e < 1.0) }' && ok || bad "portable fast call took ${fast_elapsed}s" -if [[ -s "$WATCHDOG_SLEEP_PID_FILE" ]] && kill -0 "$(cat "$WATCHDOG_SLEEP_PID_FILE")" 2>/dev/null; then - bad "portable success left watchdog sleep alive" +[[ -s "$T/fast-observed/timer.pid" ]] && ok || bad "portable success timer was not observed" +if [[ -s "$T/fast-observed/remaining-jobs" ]] || + { [[ -s "$T/fast-observed/timer.pid" ]] && kill -0 "$(cat "$T/fast-observed/timer.pid")" 2>/dev/null; }; then + bad "portable success left watchdog timer alive" else ok fi +[[ -f "$T/fast-observed/private" ]] && ok || bad "timer FIFO or directory permissions are not private" +[[ -s "$T/fast-observed/control.path" && ! -e "$(dirname "$(cat "$T/fast-observed/control.path")")" ]] && ok || bad "timer private directory was not removed" +grep -qx 'restored' "$T/fast-observed/caller-fd9" && ok || bad "timer cancellation damaged caller FD9" +[[ ! -s "$T/fast-observed/signals" ]] && ok || bad "fast timer cancellation signalled a numeric PID" + +# An ignored TERM disposition is inherited by the timer. Cancellation must +# still reap it immediately rather than waiting out the whole two-second limit. +mkdir "$T/ignored-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/ignored-observed" IGNORE_TERM=1 DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/fast-echo" CLAUDE_PROJECT_DIR="$T/project" \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/ignored.out" 2> "$T/ignored.err" +ignored_elapsed=$(cat "$T/ignored-observed/elapsed") +grep -q 'exec hello' "$T/ignored.out" && ok || bad "ignored-TERM fast output" +awk -v e="$ignored_elapsed" 'BEGIN { exit !(e < 1.0) }' && ok || bad "ignored-TERM fast call took ${ignored_elapsed}s" +[[ -s "$T/ignored-observed/timer.pid" ]] && ok || bad "ignored-TERM timer was not observed" +if [[ -s "$T/ignored-observed/remaining-jobs" ]] || + { [[ -s "$T/ignored-observed/timer.pid" ]] && kill -0 "$(cat "$T/ignored-observed/timer.pid")" 2>/dev/null; }; then + bad "ignored-TERM success left watchdog timer alive" +else + ok +fi + +mkdir "$T/early-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/early-observed" DELAY_TIMER_OPEN=1 DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/fast-echo" CLAUDE_PROJECT_DIR="$T/project" \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/early.out" 2> "$T/early.err" +early_elapsed=$(cat "$T/early-observed/elapsed") +grep -q 'exec hello' "$T/early.out" && ok || bad "early timer cancellation lost CLI output" +awk -v e="$early_elapsed" 'BEGIN { exit !(e < 1.0) }' && ok || bad "cancellation before timer read took ${early_elapsed}s" +[[ -s "$T/early-observed/timer.pid" && ! -s "$T/early-observed/remaining-jobs" ]] && ok || bad "early cancellation did not reap timer" +[[ -s "$T/early-observed/control.path" && ! -e "$(dirname "$(cat "$T/early-observed/control.path")")" ]] && ok || bad "early timer cancellation left its FIFO" +[[ -f "$T/early-observed/cancel-before-open" && -f "$T/early-observed/open-started" ]] && ok || bad "cancellation-before-timer-open seam was not exercised" + +# Cancellation after a timer has naturally exited/reaped is safe, and must +# never fall back to signalling its former numeric PID. The helper restores +# the caller's FD9 even when there is no longer another FIFO reader. +TIMER_FIXTURE_DIR="$T/exited-timer"; export TIMER_FIXTURE_DIR +mkdir -m 700 "$TIMER_FIXTURE_DIR" +mkfifo -m 600 "$TIMER_FIXTURE_DIR/control" +/bin/bash -c ' + source "$1/plugin/lib/adapter.sh" + kill() { printf "%s\n" "$*" >> "$TIMER_FIXTURE_DIR/signals"; return 99; } + exec 9>"$TIMER_FIXTURE_DIR/caller-fd9" + (trap - EXIT; if IFS= read -r -t 1 message <>"$TIMER_FIXTURE_DIR/control"; then exit 8; else printf "timeout\n" > "$TIMER_FIXTURE_DIR/deadline"; fi) /dev/null 2>&1 & + timer_pid=$! + wait "$timer_pid" + _adapter_stop_timer "$timer_pid" "$TIMER_FIXTURE_DIR/control" + printf "restored\n" >&9 + jobs -pr > "$TIMER_FIXTURE_DIR/remaining-jobs" +' _ "$ROOT" > "$T/exited.out" 2> "$T/exited.err" +grep -qx 'timeout' "$TIMER_FIXTURE_DIR/deadline" && ok || bad "timer did not reach its natural deadline" +[[ ! -s "$TIMER_FIXTURE_DIR/signals" && ! -s "$TIMER_FIXTURE_DIR/remaining-jobs" ]] && ok || bad "late timer cancellation signalled a PID or retained a job" +grep -qx 'restored' "$TIMER_FIXTURE_DIR/caller-fd9" && ok || bad "late timer cancellation damaged caller FD9" + +# Actual CLI exit status, not timer completion, determines ordinary failures. +mkdir "$T/error-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/error-observed" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/codex" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=error \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/error.out" 2> "$T/error.err" || error_rc=$? +[[ "${error_rc:-0}" -eq 1 ]] && grep -q '^CLI_ERROR:.*exit 7.*deliberate offline failure' "$T/error.err" && ok || bad "portable watchdog lost CLI failure status/classification" +[[ -s "$T/error-observed/control.path" && ! -e "$(dirname "$(cat "$T/error-observed/control.path")")" && ! -s "$T/error-observed/remaining-jobs" ]] && ok || bad "ordinary CLI failure left a timer or FIFO" + +# The invocation's EXIT cleanup must not survive its local scope and resolve +# caller globals. Preserve a prior caller EXIT hook on success and failure, +# on both the portable branch and the timeout-binary compatibility branch. +mkdir -p "$T/gnu/bin" +cat > "$T/gnu/bin/timeout" <<'EOF' +#!/usr/bin/env bash +shift +exec "$@" +EOF +chmod +x "$T/gnu/bin/timeout" +cat > "$T/caller-scope.sh" <<'EOF' +#!/usr/bin/env bash +source "$1/plugin/lib/codex-wrapper.sh" +timer_dir="$CALLER_STATE" +timer_pid="" +timer_control_file="" +trap 'printf "restored\n" > "$CALLER_STATE/prior-exit"' EXIT +if invoke_codex hello 10 2; then call_rc=0; else call_rc=$?; fi +[[ -f "$timer_dir/control" && -f "$timer_dir/deadline" && -f "$timer_dir/processes" ]] || exit 18 +exit "$call_rc" +EOF +for caller_case in portable-success gnu-success portable-failure; do + caller_state="$T/caller-$caller_case" + mkdir "$caller_state" + printf 'caller\n' > "$caller_state/control" + printf 'caller\n' > "$caller_state/deadline" + printf 'caller\n' > "$caller_state/processes" + caller_path="$T/bin:/usr/bin:/bin" caller_force=1 caller_mode=fast caller_expected=0 + if [[ "$caller_case" == "gnu-success" ]]; then caller_path="$T/gnu/bin:$caller_path"; caller_force=0; fi + if [[ "$caller_case" == "portable-failure" ]]; then caller_mode=error; caller_expected=1; fi + if HOME="$T/home" PATH="$caller_path" CALLER_STATE="$caller_state" DEVSQUAD_FORCE_PORTABLE_TIMEOUT="$caller_force" DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/codex" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE="$caller_mode" \ + /bin/bash "$T/caller-scope.sh" "$ROOT" > "$T/$caller_case.out" 2> "$T/$caller_case.err"; then caller_rc=0; else caller_rc=$?; fi + [[ "$caller_rc" -eq "$caller_expected" && -f "$caller_state/control" && -f "$caller_state/deadline" && -f "$caller_state/processes" ]] && ok || bad "$caller_case EXIT cleanup deleted caller-owned paths" + [[ -f "$caller_state/prior-exit" ]] && grep -qx restored "$caller_state/prior-exit" && ok || bad "$caller_case failed to restore caller EXIT trap" +done + +# An exit while invocation locals are still live must cancel/reap its timer +# and remove its private FIFO, rather than simply dropping the cleanup trap. +mkdir "$T/aborted-observed" +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" OBSERVATION_DIR="$T/aborted-observed" TRIGGER_POLL_EXIT=1 DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/fast-echo" CLAUDE_PROJECT_DIR="$T/project" \ + /bin/bash "$T/fast-call.sh" "$ROOT" > "$T/aborted.out" 2> "$T/aborted.err" || aborted_rc=$? +[[ "${aborted_rc:-0}" -eq 17 && -s "$T/aborted-observed/timer.pid" ]] && ok || bad "active-scope exit cleanup seam was not exercised" +if [[ -s "$T/aborted-observed/timer.pid" ]] && kill -0 "$(cat "$T/aborted-observed/timer.pid")" 2>/dev/null; then + bad "active-scope exit left watchdog timer alive" +else + ok +fi +[[ -s "$T/aborted-observed/control.path" && ! -e "$(dirname "$(cat "$T/aborted-observed/control.path")")" ]] && ok || bad "active-scope exit left private timer paths" # Model lookup remains optional when jq is absent from PATH. mkdir -p "$T/nojq" @@ -98,6 +275,22 @@ else ok fi +# A costly inspection must not extend a one-second real deadline. Keep the +# existing <4s timeout assertion, and verify the resistant root is still gone. +start=$(perl -MTime::HiRes=time -e 'printf "%.6f", time') +HOME="$T/home" PATH="$T/bin:/usr/bin:/bin" DEVSQUAD_FORCE_PORTABLE_TIMEOUT=1 DEVSQUAD_TEST_ADAPTER_EXECUTABLE="$T/bin/codex" CLAUDE_PROJECT_DIR="$T/project" FAKE_MODE=root_ignore FAKE_PROBE_DELAY=0.2 ROOT_PID_FILE="$T/slow-root.pid" ROOT_READY_FILE="$T/slow-root.ready" \ + /bin/bash -c 'source "$1/plugin/lib/codex-wrapper.sh"; invoke_codex hello 10 1' _ "$ROOT" > "$T/slow.out" 2> "$T/slow.err" || slow_rc=$? +finish=$(perl -MTime::HiRes=time -e 'printf "%.6f", time') +slow_elapsed=$(awk -v s="$start" -v f="$finish" 'BEGIN { printf "%.3f", f-s }') +[[ -f "$T/slow-root.ready" ]] && ok || bad "slow-probe root never reached readiness" +[[ "${slow_rc:-0}" -eq 1 ]] && grep -q '^TIMEOUT:' "$T/slow.err" && ok || bad "slow-probe timeout classification" +awk -v e="$slow_elapsed" 'BEGIN { exit !(e >= 0.8 && e < 4.0) }' && ok || bad "slow process probes extended one-second deadline to ${slow_elapsed}s" +if [[ -s "$T/slow-root.pid" ]] && kill -0 "$(cat "$T/slow-root.pid")" 2>/dev/null; then + bad "slow-probe timeout left TERM-ignoring root alive" +else + ok +fi + # Newline manifests preserve spaces and enumerate TSX. mkdir -p "$T/project/ui/My Folder" printf 'export const Card = 1;\n' > "$T/project/ui/My Folder/Card.tsx" @@ -142,5 +335,6 @@ PATH="$T/bin:/usr/bin:/bin" CLAUDE_PROJECT_DIR="$T/project" \ grep -q 'OUTSIDE_MARKER' "$T/ancestor.out" && bad "symlink ancestor leaked outside content" || ok grep -q 'symlink ancestor escapes project scope: safe/code.ts' "$T/ancestor.err" && ok || bad "symlink ancestor omission not reported" +echo " watchdog timing: fast=${fast_elapsed}s ignored-TERM=${ignored_elapsed}s early-cancel=${early_elapsed}s tree=${elapsed}s resistant=${root_elapsed}s slow-probe=${slow_elapsed}s" echo " m1_legacy: ${PASS} passed, ${FAIL} failed" [[ "$FAIL" -eq 0 ]] From 1c9d5dac720e926d72dfb94577c4079ca7ff20db Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 18:23:33 -0700 Subject: [PATCH 193/197] WIP checkpoint: release: retain failed3.14 public gate and bounded test synchronization repair status (2026-10-02 18:23) --- docs/plans/engineering-team/RESUME.md | 13 +++++-- .../evidence/public-release-0.11.0.json | 35 +++++++++++++++++-- 2 files changed, 44 insertions(+), 4 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index eecb0c4..5b7dc70 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -12,6 +12,13 @@ hosted service. PR1: https://github.com/joshidikshant/devsquad/pull/1. Publication is pending final public CI, merge and exact-commit release assets. The whole engineering-team plan remains incomplete. +Latest public CI37084805784 on5a576d3: Bash259/11 files (50 focused) and +optional MCP passed; Python3.14.7 FAILED486 tests/679.063s at coordinator +crash receipt recovery (fixed.6s sleep, live worker). Python3.11 still active, +not accepted. Runtime's live-process refusal is correct. Isolated terminal/ +readiness repair is limited to positively synchronizing the test's child-start, +receipt publication and actual owned runner exit; no core change. No merge/tag. + ## Accepted gates — do not repeat unchanged - Frozen core b83dda7fbf691503d3adf3c9ea6ecebd4071ba16 passed604 tests/ @@ -65,8 +72,10 @@ released run its normal scoped plugin updater and verify0.11.0. ## Exact next action -1. Root integrated Bash259/affected45/reference/diff/JSON gates pass; - checkpoint using git-safety. Core/full/native/installed gates need no repeat. +1. Root integrated watchdog Bash259/affected45/reference/diff/JSON gates pass. + Finish/review the isolated coordinator-recovery test synchronization repair, + record both CI platform outcomes, run affected/Bash checks and checkpoint. + Core source/full/native/installed gates need no unchanged repeat. 2. Push codex/engineering-team, require final exact-head public CI green: legacy Bash3.2, core Python3.11 and3.14, optional MCP. Earlier CI 37082086201/37082130929 was cancelled after true legacy failure; optional diff --git a/docs/plans/engineering-team/evidence/public-release-0.11.0.json b/docs/plans/engineering-team/evidence/public-release-0.11.0.json index 6a7a1aa..7f566f0 100644 --- a/docs/plans/engineering-team/evidence/public-release-0.11.0.json +++ b/docs/plans/engineering-team/evidence/public-release-0.11.0.json @@ -1,7 +1,7 @@ { "schema_version": 1, "recorded_on": "2026-10-02", - "status": "integrated_release_candidate_public_ci_pending", + "status": "public_ci_recovery_test_synchronization_repair_in_progress", "authorization": "User chose public GitHub release in existing joshidikshant/devsquad; no package registry or hosted service", "pull_request": "https://github.com/joshidikshant/devsquad/pull/1", "frozen_core_candidate": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", @@ -161,7 +161,7 @@ "antigravity": "agyCLI1.2.14 updated-release MCP status verified; automatic worker roles unverified and IDE out of scope", "whole_plan": "not complete" }, - "remaining": "Push final candidate; require exact-head public CI green, merge PR1 and publish v0.11.0 exact-commit verified assets; local Claude plugin normal update. Existing CodexApp connection needs scoped reconnect. Core604/native/installed9SDK/agyCLI gates already passed; do not repeat unchanged. Whole-plan/Council/desktop residuals remain separate.", + "remaining": "Finish independently reviewed recovery-test synchronization repair; preserve failure record; root affected/Bash checks and exact-head public CI green, merge PR1 and publish exact-commit assets. Existing CodexApp connection needs scoped reconnect; accepted core/native/install/agy proofs remain source-equivalent, not substitute for failed public CI.", "integrated_legacy_gate": { "source_checkpoint": "3fdc6bd065c577a9ed59566573fdce399e9b585a", "adapter_blob": "1847357c19677f276b06886386e2ac93e586b203", @@ -188,5 +188,36 @@ "reference": "passed", "core_tree_unchanged": true, "failure_history": "evidence/legacy-watchdog-release-repair.json" + }, + "final_ci_attempt": { + "run_id": 37084805784, + "url": "https://github.com/joshidikshant/devsquad/actions/runs/37084805784", + "head": "5a576d39469c198a3cc93474960886b875363419", + "legacy": "passed259/11 files;50 focused", + "legacy_timings_seconds": { + "fast": 0.334, + "ignored_TERM": 0.335, + "pre_open_cancel": 0.221, + "TERM_resistant": 3.609, + "slow_ps": 1.518 + }, + "optional_mcp": "passed", + "python314": { + "version": "3.14.7", + "tests": 486, + "seconds": 679.063, + "failures": 1, + "errors": 0, + "skips": 2, + "unraisable": [], + "case": "test_coordinator_crash_imports_runner_receipt_once", + "assertion": "live not in succeeded/already_finalized", + "location": "test/core/test_service.py:1108", + "status": "failed_not_accepted" + }, + "python311": "still_running_not_accepted", + "duplicate_push_run": 37084802403, + "duplicate_push_status": "cancelled_to_avoid_duplicate_CI", + "diagnosis": "Fixed.6s sleep does not establish child-start/runner-exit boundary; running state precedes runner gate. Live refusal is correct ownership fence. Isolated bounded positive synchronization repair and independent review underway; no core runtime edit authorized from this test observation." } } From 2d5a6f9050374cd58fa5bb5bdd393d0ab7cf282b Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 18:33:56 -0700 Subject: [PATCH 194/197] WIP checkpoint: release: preserve receipt synchronization and truthful neighboring waiter failure (2026-10-02 18:33) --- docs/plans/engineering-team/RESUME.md | 14 +++- .../evidence/public-release-0.11.0.json | 32 +++++++++- .../release-receipt-synchronization.json | 64 +++++++++++++++++++ test/core/test_service.py | 34 +++++++++- 4 files changed, 136 insertions(+), 8 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/release-receipt-synchronization.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 5b7dc70..dd1e5fe 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -14,10 +14,18 @@ The whole engineering-team plan remains incomplete. Latest public CI37084805784 on5a576d3: Bash259/11 files (50 focused) and optional MCP passed; Python3.14.7 FAILED486 tests/679.063s at coordinator -crash receipt recovery (fixed.6s sleep, live worker). Python3.11 still active, -not accepted. Runtime's live-process refusal is correct. Isolated terminal/ +crash receipt recovery (fixed.6s sleep, live worker). Python3.11 PASSED604/ +1013.419s, two SDK skips/no errors/failures/unraisable on that earlier head. +Runtime's live-process refusal is correct. Isolated terminal/ readiness repair is limited to positively synchronizing the test's child-start, -receipt publication and actual owned runner exit; no core change. No merge/tag. +receipt publication and actual owned runner exit; no core change. Controlled +original red and synchronized green proven on3.12.14/local3.14.6 (not CI.7). +edfcf65 test review clean/agent33 module each3.12/3.14/Bash259 pass. +Root integration Bash259 passed;33 service tests FAILED1/32.435s at the +neighboring before-gate crash waiter598 (ambiguous !=dead). Concurrent Bash +alone is not causal proof. Isolated test-only exactdead wait repair retains +the5s boundary and all before-gate/exact-once assertions; no core edit. +Independent review/root rerun pending. No merge/tag. ## Accepted gates — do not repeat unchanged diff --git a/docs/plans/engineering-team/evidence/public-release-0.11.0.json b/docs/plans/engineering-team/evidence/public-release-0.11.0.json index 7f566f0..89ae06a 100644 --- a/docs/plans/engineering-team/evidence/public-release-0.11.0.json +++ b/docs/plans/engineering-team/evidence/public-release-0.11.0.json @@ -161,7 +161,7 @@ "antigravity": "agyCLI1.2.14 updated-release MCP status verified; automatic worker roles unverified and IDE out of scope", "whole_plan": "not complete" }, - "remaining": "Finish independently reviewed recovery-test synchronization repair; preserve failure record; root affected/Bash checks and exact-head public CI green, merge PR1 and publish exact-commit assets. Existing CodexApp connection needs scoped reconnect; accepted core/native/install/agy proofs remain source-equivalent, not substitute for failed public CI.", + "remaining": "Finish independently reviewed before-gate exit waiter test correction and root affected/Bash gates. Retain both public3.14 and root-service failed attempts; then exact-head CI green, PR1 merge and exact-commit release assets. Core/runtime native/install/agy proof unchanged; old existing CodexApp connection still requires scoped reconnect.", "integrated_legacy_gate": { "source_checkpoint": "3fdc6bd065c577a9ed59566573fdce399e9b585a", "adapter_blob": "1847357c19677f276b06886386e2ac93e586b203", @@ -215,9 +215,35 @@ "location": "test/core/test_service.py:1108", "status": "failed_not_accepted" }, - "python311": "still_running_not_accepted", + "python311": { + "tests": 604, + "seconds": 1013.419, + "failures": 0, + "errors": 0, + "skips": 2, + "unraisable": [], + "status": "passed_on5a_prior_to_test_synchronization_repair" + }, "duplicate_push_run": 37084802403, "duplicate_push_status": "cancelled_to_avoid_duplicate_CI", - "diagnosis": "Fixed.6s sleep does not establish child-start/runner-exit boundary; running state precedes runner gate. Live refusal is correct ownership fence. Isolated bounded positive synchronization repair and independent review underway; no core runtime edit authorized from this test observation." + "diagnosis": "Scheduling-only defect proven on exact5a controlled original1.2s fake child: original method failed same live assertion on3.12.14/3.14.6; synchronized test passes same controlled case. Product live ownership fence remains unchanged. Independent frozen patch review clean; affected module/Bash checkpoint pending.", + "status": "completed_failure;3.14 recovery-test timing assumption failed, other three jobs passed" + }, + "root_service_integration_attempt": { + "source_checkpoint": "edfcf65b92eeee97323b20db6d252b18e2fbc99a", + "test_blob": "e19495c811c5e52e3d66e5588d18f325e306a8ad", + "bash": "passed259/11files", + "service": { + "tests": 33, + "seconds": 32.435, + "failures": 1, + "case": "test_crash_after_runner_identity_before_gate_requeues_without_execution", + "location": "test/core/test_service.py:598", + "assertion": "ambiguous != dead", + "status": "failed_not_accepted" + }, + "concurrency": "Bash and service module ran concurrently on same host; no causal attribution asserted from concurrency alone", + "diagnosis": "Existing neighboring waiter stops at not-live yet assertsdead. Transient process identity ambiguity during exit is conservatively valid; bounded exactdead synchronization repair in isolated test only, independent review pending.", + "runtime_core_unchanged": true } } diff --git a/docs/plans/engineering-team/evidence/release-receipt-synchronization.json b/docs/plans/engineering-team/evidence/release-receipt-synchronization.json new file mode 100644 index 0000000..48ad16a --- /dev/null +++ b/docs/plans/engineering-team/evidence/release-receipt-synchronization.json @@ -0,0 +1,64 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "Test-only coordinator-crash receipt synchronization; no recovery/product changes or whole-suite rerun.", + "baseline_revision": "5a576d39469c198a3cc93474960886b875363419", + "branch": "codex/release-receipt-sync", + "preserved_branch": "codex/release-watchdog", + "test_blob": "e19495c811c5e52e3d66e5588d18f325e306a8ad", + "test_diff_sha256": "ef3c84d775695a832fafc6681e494262ebdfdd339be36170c4b6d44ee451facb", + "public_CI_failure": { + "run_id": "37084805784", + "head_sha": "5a576d39469c198a3cc93474960886b875363419", + "reported_by": "root", + "python": "3.14.7", + "tests_before_failfast_stop": 486, + "seconds": 679.063, + "test": "test_service.ServiceTest.test_coordinator_crash_imports_runner_receipt_once", + "assertion": "first disposition was live, expected succeeded/already_finalized at original test_service.py:1108", + "other_jobs": "Bash and optional MCP passed; Python3.11 completed604 tests/1013.419s with2 optional SDK skips, zero failures/errors/unraisable. This agent did not cancel any job." + }, + "cause": "The test assumed a .6s sleep after killing the coordinator implied durable runner completion. Actual gated Python startups, child work, drain/fsync and receipt publication need not fit that delay. Product import_durable correctly returns live before receipt parsing. Also, running is committed before the runner gate is released; child_record is the positive started-child phase boundary.", + "controlled_red": { + "method": "Run the original exact test with only its existing internal fake child's .3s work delay replaced by1.2s through a diagnostic Service.start wrapper. Capture the real resume response, original runner identity, receipt presence and attempt count. Before fixture teardown, positively wait for owned runner death and receipt publication.", + "python312": {"version": "3.12.14", "tests": 1, "failures": 1, "seconds": 1.85}, + "python314": {"version": "3.14.6", "tests": 1, "failures": 1, "seconds": 2.023}, + "same_observation_both_versions": {"disposition": "live", "runner_identity": "live", "receipt_present": false, "attempt_count": 1}, + "cleanup_both_versions": {"runner_identity": "dead", "receipt_present": true}, + "uncontrolled_baseline": "The original test also passed in isolation on3.12.14/.991s and3.14.6/1.052s, consistent with a scheduling-sensitive assumption rather than a deterministic import defect." + }, + "repair": { + "changed_code": "Only test/core/test_service.py:32 inserted/2 removed lines in the existing method.", + "started_phase": "Wait for the original attempt's durable child_record before killing the coordinator, so this is not before-gate/unstarted recovery.", + "import_phase": "Replace .6s sleep with a bounded wait for the same original receipt path and inspect_process(pid,pgid,start_id) == dead; assert both before the first import.", + "bounds": "Original .3s fixture work is unchanged. The5s positive wait boundary matches adjacent receipt/import tests. No product timeout or timing assertion was widened.", + "live_fence": "Existing test_dead_supervisor_with_live_child_never_relaunches is unchanged and included in both complete service-module gates. No new short-delay live assertion or recovery requeue/takeover was introduced.", + "exact_once": "Original successful disposition and repeated terminal resume ConflictError retained. Require one attempt, unchanged id/token/pid/pgid/start, finished status, exactly3 artifacts and exactly one attempt.output/run.succeeded event." + }, + "controlled_green": { + "method": "Repeat the identical1.2s delayed-child diagnostic against the repaired test.", + "python312": {"version": "3.12.14", "tests": 1, "failures": 0, "seconds": 1.72}, + "python314": {"version": "3.14.6", "tests": 1, "failures": 0, "seconds": 1.867}, + "same_observation_both_versions": {"disposition": "succeeded", "runner_identity": "dead", "receipt_present": true, "attempt_count": 1} + }, + "affected_service_gates": { + "environment": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core", + "command": "INTERPRETER -m unittest test_service -q", + "python312": {"interpreter": "/Users/Dikshant/.cache/codex-runtimes/codex-primary-runtime/dependencies/python/bin/python3", "version": "3.12.14", "tests": 33, "seconds": 36.966, "failures": 0, "errors": 0, "skips": 0}, + "python314": {"interpreter": "/opt/homebrew/bin/python3.14", "version": "3.14.6", "tests": 33, "seconds": 39.543, "failures": 0, "errors": 0, "skips": 0}, + "coverage": "Complete service module, including live refusal, before-gate crash, orphan cancellation, importer-finalization crash, coordinator receipt recovery and concurrent exact-once receipt import." + }, + "independent_review": { + "reviewer": "r6_readiness", + "scope": "Frozen changed test and read-only recovery semantics; no edits/full/provider/install actions.", + "result": "clean; scheduling defect, live-process and exact-once fences preserved", + "focused_tests": ["test_coordinator_crash_imports_runner_receipt_once", "test_importer_crash_after_artifact_finalize_has_no_false_completion", "test_two_processes_import_one_runner_receipt_atomically"], + "python312": {"version": "3.12.14", "tests": 3, "seconds": 3.415, "failures": 0}, + "python314": {"version": "3.14.6", "tests": 3, "seconds": 3.161, "failures": 0}, + "before_after_hashes_unchanged": true + }, + "core_tree_unchanged": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "other_gates": "Generated reference and diff checks pass; required Bash259 gate is run before the checkpoint.", + "limitations": "Local3.14 is3.14.6, not CI3.14.7. No local full604 suite, provider/native request, install, publication, global settings or root source/docs change. Final exact-head public CI remains required and root-owned.", + "next_action": "Root integrates the narrow checkpoint, records final gate/checkpoint identifiers in authoritative release evidence, verifies its bounded integration gate and pushes the next exact public CI candidate." +} diff --git a/test/core/test_service.py b/test/core/test_service.py index 8a9420f..e19495c 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -1100,16 +1100,46 @@ def test_coordinator_crash_imports_runner_receipt_once(self): self.wait_state(started["run_id"],{"running"}) store=Store(self.runtime/"state.sqlite3",self.runtime/"artifacts") try: + attempt=store.attempt(started["run_id"]) + # Running is committed before the runner gate is released. Prove + # the child actually started before exercising coordinator death. + child_record=Path(attempt["child_record"]) + deadline=time.monotonic()+5 + while not child_record.is_file() and time.monotonic() Date: Fri, 2 Oct 2026 18:39:53 -0700 Subject: [PATCH 195/197] WIP checkpoint: release: correct crash-phase test synchronization without changing runtime (2026-10-02 18:39) --- docs/plans/engineering-team/RESUME.md | 12 +++- .../evidence/public-release-0.11.0.json | 34 +++++++++- .../release-runner-exit-synchronization.json | 68 +++++++++++++++++++ test/core/test_service.py | 4 +- 4 files changed, 112 insertions(+), 6 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/release-runner-exit-synchronization.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index dd1e5fe..2fa1fd5 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -25,7 +25,11 @@ Root integration Bash259 passed;33 service tests FAILED1/32.435s at the neighboring before-gate crash waiter598 (ambiguous !=dead). Concurrent Bash alone is not causal proof. Isolated test-only exactdead wait repair retains the5s boundary and all before-gate/exact-once assertions; no core edit. -Independent review/root rerun pending. No merge/tag. +e7b63b0 confirmed-dead waiter is integrated: both independent reviews clean, +same5s boundary/core unchanged, agent33 service tests3.12/3.14 pass; +root complete33 module PASSED29.820s/no failures/errors/skips. Root Bash259/ +11files/reference/JSON/diff passed. Corrected checkpoint/push/new exact-head +public CI pending. No merge/tag. ## Accepted gates — do not repeat unchanged @@ -81,8 +85,8 @@ released run its normal scoped plugin updater and verify0.11.0. ## Exact next action 1. Root integrated watchdog Bash259/affected45/reference/diff/JSON gates pass. - Finish/review the isolated coordinator-recovery test synchronization repair, - record both CI platform outcomes, run affected/Bash checks and checkpoint. + Both receipt/dead-wait test repairs are integrated/reviewed; root33 passes. + Finish Bash/JSON/reference/diff checks and checkpoint the corrected candidate. Core source/full/native/installed gates need no unchanged repeat. 2. Push codex/engineering-team, require final exact-head public CI green: legacy Bash3.2, core Python3.11 and3.14, optional MCP. Earlier CI @@ -110,5 +114,7 @@ change. Raw provider diagnostics/credentials remain outside tracked evidence. Evidence: evidence/public-release-0.11.0.json, evidence/legacy-watchdog-release-repair.json, R6-terminal-readiness-partial, +evidence/release-receipt-synchronization.json and +evidence/release-runner-exit-synchronization.json, R7 reconciliation/final-network diagnostics, R8-installed-workflows and R3/R4/R5 closure files. Historical recovery detail is preserved in Git. diff --git a/docs/plans/engineering-team/evidence/public-release-0.11.0.json b/docs/plans/engineering-team/evidence/public-release-0.11.0.json index 89ae06a..03d166e 100644 --- a/docs/plans/engineering-team/evidence/public-release-0.11.0.json +++ b/docs/plans/engineering-team/evidence/public-release-0.11.0.json @@ -1,7 +1,7 @@ { "schema_version": 1, "recorded_on": "2026-10-02", - "status": "public_ci_recovery_test_synchronization_repair_in_progress", + "status": "corrected_test_candidate_root_service_passed_public_ci_pending", "authorization": "User chose public GitHub release in existing joshidikshant/devsquad; no package registry or hosted service", "pull_request": "https://github.com/joshidikshant/devsquad/pull/1", "frozen_core_candidate": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", @@ -161,7 +161,7 @@ "antigravity": "agyCLI1.2.14 updated-release MCP status verified; automatic worker roles unverified and IDE out of scope", "whole_plan": "not complete" }, - "remaining": "Finish independently reviewed before-gate exit waiter test correction and root affected/Bash gates. Retain both public3.14 and root-service failed attempts; then exact-head CI green, PR1 merge and exact-commit release assets. Core/runtime native/install/agy proof unchanged; old existing CodexApp connection still requires scoped reconnect.", + "remaining": "Complete final Bash/JSON/reference/diff checkpoint and push corrected candidate; require all four exact-head public CI jobs green, merge PR1, publish v0.11.0 exact-merge verified assets and scoped local Claude plugin update. Existing CodexApp DevSquad MCP reconnect and original Council/desktop residuals remain separate.", "integrated_legacy_gate": { "source_checkpoint": "3fdc6bd065c577a9ed59566573fdce399e9b585a", "adapter_blob": "1847357c19677f276b06886386e2ac93e586b203", @@ -245,5 +245,35 @@ "concurrency": "Bash and service module ran concurrently on same host; no causal attribution asserted from concurrency alone", "diagnosis": "Existing neighboring waiter stops at not-live yet assertsdead. Transient process identity ambiguity during exit is conservatively valid; bounded exactdead synchronization repair in isolated test only, independent review pending.", "runtime_core_unchanged": true + }, + "corrected_service_gate": { + "receipt_checkpoint": "edfcf65b92eeee97323b20db6d252b18e2fbc99a", + "runner_exit_checkpoint": "e7b63b0d52808fdac109707abf9851ae9f3dabde", + "test_blob": "5456e21d10f7e65b06445a266ac0b177d8bd2066", + "core_runtime_unchanged": true, + "independent_reviews": "clean; both committed source-equivalent to reviewed candidates", + "agent_service": { + "python312_tests": 33, + "python312_seconds": 32.621, + "python314_tests": 33, + "python314_seconds": 35.47, + "python314_version": "3.14.6_not_CI3.14.7", + "failures": 0, + "errors": 0, + "skips": 0 + }, + "root_service": { + "tests": 33, + "seconds": 29.82, + "failures": 0, + "errors": 0, + "skips": 0 + }, + "root_bash": {"assertions":259,"files":11,"failures":0}, + "evidence": [ + "evidence/release-receipt-synchronization.json", + "evidence/release-runner-exit-synchronization.json" + ], + "bounds": "Original .3s work and5s positive boundaries unchanged; confirmed child-start/receipt/original runner dead, no live/ambiguous authority relaxation; exactonce retained" } } diff --git a/docs/plans/engineering-team/evidence/release-runner-exit-synchronization.json b/docs/plans/engineering-team/evidence/release-runner-exit-synchronization.json new file mode 100644 index 0000000..aeae57d --- /dev/null +++ b/docs/plans/engineering-team/evidence/release-runner-exit-synchronization.json @@ -0,0 +1,68 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "Test-only positive runner-death synchronization in the before-gate crash case; no core or recovery changes.", + "baseline_revision": "edfcf65b92eeee97323b20db6d252b18e2fbc99a", + "branch": "codex/release-runner-exit-sync", + "preserved_branch": "codex/release-receipt-sync", + "test_blob": "5456e21d10f7e65b06445a266ac0b177d8bd2066", + "test_diff_sha256": "e20bc04a593e35a9d5612591c19ab54081f974e3ba57256c3ba927007de0f7b0", + "root_integration_failure": { + "reported_by": "root", + "tests": 33, + "seconds": 32.435, + "failures": 1, + "test": "test_service.ServiceTest.test_crash_after_runner_identity_before_gate_requeues_without_execution", + "assertion": "ambiguous != dead at original test_service.py:598", + "context": "Root ran the complete service module while Bash was also active after integrating edfc. Bash259 passed. This agent did not repeat or edit that root gate." + }, + "cause": "The existing waiter continued only while inspect_process returned live. A dying runner can legitimately be ambiguous when its start identity was observed but getpgid then loses the PID, or the group inventory is inconclusive. Ambiguous is not confirmed dead. The test exited its waiter on that intermediate classification, then asserted dead. The product classifier correctly fails closed and is unchanged.", + "controlled_diagnostic": { + "scope": "Only the test_service.inspect_process alias is wrapped; devsquad.supervisor.inspect_process and platform calls are untouched.", + "method": "For the same captured pid/pgid/start identity, delegate real inspection first. Preserve actual live classifications. At the first genuine dead boundary, return ambiguous for two successive observations, then delegate genuine classifications. Two ambiguous observations exercise both the original loop condition and its final assertion without permitting early recovery.", + "trace_both_versions": ["live", "live", "live", "live", "ambiguous", "ambiguous"], + "identity_binding": "Every observation asserts the original pid/pgid/start tuple is unchanged. The injected observations each follow an actual dead classification, so the owned runner is already gone before any failed-test fixture teardown.", + "original_red": { + "python312": {"version": "3.12.14", "tests": 1, "failures": 1, "errors": 0, "seconds": 0.445}, + "python314": {"version": "3.14.6", "tests": 1, "failures": 1, "errors": 0, "seconds": 0.458}, + "assertion": "ambiguous != dead at598 on both versions" + }, + "corrected_green": { + "python312": {"version": "3.12.14", "tests": 1, "failures": 0, "errors": 0, "seconds": 0.856}, + "python314": {"version": "3.14.6", "tests": 1, "failures": 0, "errors": 0, "seconds": 0.946}, + "final_trace": ["dead", "dead"] + } + }, + "repair": { + "changed_code": "Only3 inserted/1 removed lines in the existing method: change while == live to while != dead and add two explanatory comments.", + "bounds": "The original5s monotonic deadline and .02s polling interval are unchanged. No sleep-only widening or product timeout change.", + "positive_boundary": "Require an actual dead classification of the original attempt identity before continuing; the final dead assertion is unchanged. Persistent ambiguous/live ownership still fails this test instead of being accepted as death.", + "preserved_assertions": "Coordinator exits24 before releasing the runner gate; original child_record and execution marker remain absent; resume launches one continuation, succeeds, never executes the blocked command, records exactly recovery_required then finished attempts and exactly one run.unstarted_attempt_recovered event. Existing live-runner refusal and receipt exact-once tests remain unchanged." + }, + "affected_service_gates": { + "environment": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core", + "command": "INTERPRETER -B -m unittest test_service -q", + "python312": {"version": "3.12.14", "tests": 33, "seconds": 32.621, "failures": 0, "errors": 0, "skips": 0}, + "python314": {"version": "3.14.6", "tests": 33, "seconds": 35.470, "failures": 0, "errors": 0, "skips": 0}, + "coverage": "Complete service module, including unstarted recovery, live-owner refusal, receipt import, orphan cancellation and concurrent exact-once import." + }, + "independent_review": { + "reviewer": "r6_readiness", + "scope": "Frozen test diff and classifier semantics; read-only offline focused checks, no core edits.", + "source_review": "clean; same bound and all final death, gate absence and exact requeue assertions retained", + "focused_tests": ["test_crash_after_runner_identity_before_gate_requeues_without_execution", "test_dead_supervisor_with_live_child_never_relaunches", "test_coordinator_crash_imports_runner_receipt_once"], + "python312": {"version": "3.12.14", "tests": 3, "seconds": 2.816, "failures": 0, "errors": 0}, + "python314": {"version": "3.14.6", "tests": 3, "seconds": 3.515, "failures": 0, "errors": 0}, + "independent_controlled_green": {"python312_seconds": 0.988, "python314_seconds": 0.947, "tests_each": 1, "failures": 0}, + "result": "clean; real ownership/refusal/receipt cases and exact alias-only diagnostic pass, with exactly two transient ambiguity observations followed by two genuine dead observations", + "before_after_hashes_unchanged": true + }, + "core_tree_unchanged": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "other_gates": { + "bash": {"command": "PYTHONDONTWRITEBYTECODE=1 /bin/bash test/run.sh", "test_files": 11, "assertions": 259, "failures": 0, "legacy_assertions": 50}, + "watchdog_seconds": {"fast": 0.276, "ignored_TERM": 0.214, "early_cancel": 0.213, "tree": 1, "resistant": 3.258, "slow_probe": 1.26}, + "generated_reference": "current", + "JSON_and_diff": "pass" + }, + "limitations": "Local3.14 is3.14.6, not public CI3.14.7. No full604 suite, provider/native request, installation, publication, global settings or root source/docs action. Root owns integration and the next exact-head public CI candidate." +} diff --git a/test/core/test_service.py b/test/core/test_service.py index e19495c..5456e21 100644 --- a/test/core/test_service.py +++ b/test/core/test_service.py @@ -591,9 +591,11 @@ def test_crash_after_runner_identity_before_gate_requeues_without_execution(self try: attempt=store.attempt(started["run_id"]) deadline=time.monotonic()+5 + # A dying runner may be ambiguous between identity and pgid lookup. + # Recovery needs confirmed death, not merely an observation of not-live. while (inspect_process( attempt["pid"],attempt["pgid"],attempt["process_start_id"], - )=="live" and time.monotonic() Date: Fri, 2 Oct 2026 18:47:46 -0700 Subject: [PATCH 196/197] WIP checkpoint: release: preserve corrected CI and delivery blocked-phase fixture finding (2026-10-02 18:47) --- docs/plans/engineering-team/RESUME.md | 15 ++++++++ .../evidence/public-release-0.11.0.json | 36 +++++++++++++++++-- 2 files changed, 48 insertions(+), 3 deletions(-) diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 2fa1fd5..230f4b9 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -31,6 +31,21 @@ root complete33 module PASSED29.820s/no failures/errors/skips. Root Bash259/ 11files/reference/JSON/diff passed. Corrected checkpoint/push/new exact-head public CI pending. No merge/tag. +Corrected remote candidate f40c618caefd100637c4686660673e159899ecb7 is now +pushed; PR CI37086958301: Bash/MCP passed, Python3.11.9 FAILED137/191.313s +at delivery repair recovery1001 (ownership_ambiguous !=retain_ownership). +Python3.14 still active, not accepted; duplicate push37086954389 cancelled. +Freeze this exact remote candidate. Any later local docs-only checkpoint is +recovery metadata, not a change to the tested/publication target. Root must +merge/tag the remotely tested f40 candidate, then record published outcomes. + +New bounded isolated repair: test SIGKILLs attempt_runner, not coordinator, +then assumes.1s implies blocked. Correct running import returns safe +ownership_ambiguous before typed blocked recovery. Positively wait for the +same dead runner and blocked/current attempt ownership_ambiguous before one +explicit retain request. Preserve exactly3 attempts/no new writer/launchedfalse/ +source/cancel assertions; no core edit. Independent review pending. No merge. + ## Accepted gates — do not repeat unchanged - Frozen core b83dda7fbf691503d3adf3c9ea6ecebd4071ba16 passed604 tests/ diff --git a/docs/plans/engineering-team/evidence/public-release-0.11.0.json b/docs/plans/engineering-team/evidence/public-release-0.11.0.json index 03d166e..0505a65 100644 --- a/docs/plans/engineering-team/evidence/public-release-0.11.0.json +++ b/docs/plans/engineering-team/evidence/public-release-0.11.0.json @@ -1,7 +1,7 @@ { "schema_version": 1, "recorded_on": "2026-10-02", - "status": "corrected_test_candidate_root_service_passed_public_ci_pending", + "status": "public_ci_delivery_blocked_phase_test_repair_in_progress", "authorization": "User chose public GitHub release in existing joshidikshant/devsquad; no package registry or hosted service", "pull_request": "https://github.com/joshidikshant/devsquad/pull/1", "frozen_core_candidate": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", @@ -161,7 +161,7 @@ "antigravity": "agyCLI1.2.14 updated-release MCP status verified; automatic worker roles unverified and IDE out of scope", "whole_plan": "not complete" }, - "remaining": "Complete final Bash/JSON/reference/diff checkpoint and push corrected candidate; require all four exact-head public CI jobs green, merge PR1, publish v0.11.0 exact-merge verified assets and scoped local Claude plugin update. Existing CodexApp DevSquad MCP reconnect and original Council/desktop residuals remain separate.", + "remaining": "Finish independently reviewed delivery blocked-phase fixture correction and relevant/root Bash gates, retain this failure. Require final public platform verification, merge PR1 and publish verified exact-merge assets; current runtime/native/install/agy proof source-equivalent and unchanged. No more Council/provider attempts.", "integrated_legacy_gate": { "source_checkpoint": "3fdc6bd065c577a9ed59566573fdce399e9b585a", "adapter_blob": "1847357c19677f276b06886386e2ac93e586b203", @@ -269,11 +269,41 @@ "errors": 0, "skips": 0 }, - "root_bash": {"assertions":259,"files":11,"failures":0}, + "root_bash": { + "assertions": 259, + "files": 11, + "failures": 0 + }, "evidence": [ "evidence/release-receipt-synchronization.json", "evidence/release-runner-exit-synchronization.json" ], "bounds": "Original .3s work and5s positive boundaries unchanged; confirmed child-start/receipt/original runner dead, no live/ambiguous authority relaxation; exactonce retained" + }, + "corrected_public_ci": { + "run_id": 37086958301, + "url": "https://github.com/joshidikshant/devsquad/actions/runs/37086958301", + "head": "f40c618caefd100637c4686660673e159899ecb7", + "status": "failed_python311;python314_still_running_not_accepted", + "duplicate_push_run": 37086954389, + "duplicate_push_status": "cancelled", + "core_runtime_and_plugin_payloads_unchanged": true, + "legacy": "passed", + "optional_mcp": "passed", + "python311": { + "version": "3.11.9", + "job_id": 111099154149, + "tests": 137, + "seconds": 191.313, + "failures": 1, + "errors": 0, + "skips": 0, + "unraisable": [], + "case": "test_killed_repair_supervisor_never_launches_a_duplicate_writer", + "location": "test/core/test_delivery_workflow.py:1001", + "assertion": "ownership_ambiguous != retain_ownership", + "status": "failed_not_accepted" + }, + "diagnosis": "Test kills attempt_runner, not live detached coordinator. Fixed.1s sleep does not establish coordinator's blocked/ownership_ambiguous projection. Running resume correctly imports/blocks and returns ownership_ambiguous; typed retain applies only once blocked. Positive phase/dead-original-runner/same-attempt synchronization repair in isolated test only; independent review underway. No runtime/source edit from this observation." } } From 8488ed44ee30c7fa5b2c20370827093715130d36 Mon Sep 17 00:00:00 2001 From: Dikshant Date: Fri, 2 Oct 2026 19:06:18 -0700 Subject: [PATCH 197/197] WIP checkpoint: release: synchronize delivery recovery fixture with actual blocked ownership phase (2026-10-02 19:06) --- docs/plans/engineering-team/RESUME.md | 26 ++++-- .../evidence/public-release-0.11.0.json | 66 +++++++++++++- ...ease-repair-ownership-synchronization.json | 88 +++++++++++++++++++ test/core/test_delivery_workflow.py | 26 +++++- 4 files changed, 195 insertions(+), 11 deletions(-) create mode 100644 docs/plans/engineering-team/evidence/release-repair-ownership-synchronization.json diff --git a/docs/plans/engineering-team/RESUME.md b/docs/plans/engineering-team/RESUME.md index 230f4b9..495acb3 100644 --- a/docs/plans/engineering-team/RESUME.md +++ b/docs/plans/engineering-team/RESUME.md @@ -34,17 +34,31 @@ public CI pending. No merge/tag. Corrected remote candidate f40c618caefd100637c4686660673e159899ecb7 is now pushed; PR CI37086958301: Bash/MCP passed, Python3.11.9 FAILED137/191.313s at delivery repair recovery1001 (ownership_ambiguous !=retain_ownership). -Python3.14 still active, not accepted; duplicate push37086954389 cancelled. -Freeze this exact remote candidate. Any later local docs-only checkpoint is -recovery metadata, not a change to the tested/publication target. Root must -merge/tag the remotely tested f40 candidate, then record published outcomes. +Python3.14.7 passed604/697.362s/twoSDKskips/no failures/errors/unraisable; +whole f40 workflow failed311 and is not accepted. Duplicate push cancelled. +f40 is rejected as a publication candidate by that failed311 job. Retain it +as source baseline only. Next integrate the narrow delivery fixture repair, +push a new exact candidate and require final public gates before merge/tag. +Later docs-only checkpoints are recovery metadata, not publication targets. New bounded isolated repair: test SIGKILLs attempt_runner, not coordinator, then assumes.1s implies blocked. Correct running import returns safe ownership_ambiguous before typed blocked recovery. Positively wait for the same dead runner and blocked/current attempt ownership_ambiguous before one explicit retain request. Preserve exactly3 attempts/no new writer/launchedfalse/ -source/cancel assertions; no core edit. Independent review pending. No merge. +source/cancel assertions; no core edit.893490a phasefix is now integrated, +exact7723f6ce reviewed blob. Agent52 delivery/service tests pass each3.12/3.14; +root19 delivery tests48.911s pass; unchanged service33/29.820s already passed. +Final root Bash passed259 assertions/11files (50 focused), reference/JSON/ +staged+unstaged diff checks passed. Checkpoint/push and exact-head public CI +remain. No merge. + +Rejected private diagnostic RELEASE+cancel cleanup overlapped import and +raised stale-phase ConflictError. Independent read-only triage: fail-closed +completion fence, no duplicate-writer/false-cancel evidence. No error-time +ledger snapshot proves ordering/final rows/lost-intent liveness; retain as +diagnostic concurrency-conflict / possible follow-up, not confirmed defect. +Corrected helper waits import DONE before teardown; no product edit/probes. ## Accepted gates — do not repeat unchanged @@ -101,7 +115,7 @@ released run its normal scoped plugin updater and verify0.11.0. 1. Root integrated watchdog Bash259/affected45/reference/diff/JSON gates pass. Both receipt/dead-wait test repairs are integrated/reviewed; root33 passes. - Finish Bash/JSON/reference/diff checks and checkpoint the corrected candidate. + Final Bash/JSON/reference/diff checks passed; checkpoint the corrected candidate. Core source/full/native/installed gates need no unchanged repeat. 2. Push codex/engineering-team, require final exact-head public CI green: legacy Bash3.2, core Python3.11 and3.14, optional MCP. Earlier CI diff --git a/docs/plans/engineering-team/evidence/public-release-0.11.0.json b/docs/plans/engineering-team/evidence/public-release-0.11.0.json index 0505a65..7884de0 100644 --- a/docs/plans/engineering-team/evidence/public-release-0.11.0.json +++ b/docs/plans/engineering-team/evidence/public-release-0.11.0.json @@ -1,7 +1,7 @@ { "schema_version": 1, "recorded_on": "2026-10-02", - "status": "public_ci_delivery_blocked_phase_test_repair_in_progress", + "status": "delivery_phase_corrected_root_gate_passed_next_public_ci_pending", "authorization": "User chose public GitHub release in existing joshidikshant/devsquad; no package registry or hosted service", "pull_request": "https://github.com/joshidikshant/devsquad/pull/1", "frozen_core_candidate": "b83dda7fbf691503d3adf3c9ea6ecebd4071ba16", @@ -161,7 +161,7 @@ "antigravity": "agyCLI1.2.14 updated-release MCP status verified; automatic worker roles unverified and IDE out of scope", "whole_plan": "not complete" }, - "remaining": "Finish independently reviewed delivery blocked-phase fixture correction and relevant/root Bash gates, retain this failure. Require final public platform verification, merge PR1 and publish verified exact-merge assets; current runtime/native/install/agy proof source-equivalent and unchanged. No more Council/provider attempts.", + "remaining": "Final Bash/JSON/reference/diff checks passed. Checkpoint and push delivery-test-corrected candidate. Require all final exact-head public CI jobs green; merge PR1, publish exact-merge verified v0.11.0 assets, scoped local Claude plugin update. Preserve unconfirmed diagnostic concurrency follow-up plus original Council/desktop/MCP reconnect residuals separately.", "integrated_legacy_gate": { "source_checkpoint": "3fdc6bd065c577a9ed59566573fdce399e9b585a", "adapter_blob": "1847357c19677f276b06886386e2ac93e586b203", @@ -284,7 +284,7 @@ "run_id": 37086958301, "url": "https://github.com/joshidikshant/devsquad/actions/runs/37086958301", "head": "f40c618caefd100637c4686660673e159899ecb7", - "status": "failed_python311;python314_still_running_not_accepted", + "status": "completed_failed_python311;python314_passed604", "duplicate_push_run": 37086954389, "duplicate_push_status": "cancelled", "core_runtime_and_plugin_payloads_unchanged": true, @@ -304,6 +304,64 @@ "assertion": "ownership_ambiguous != retain_ownership", "status": "failed_not_accepted" }, - "diagnosis": "Test kills attempt_runner, not live detached coordinator. Fixed.1s sleep does not establish coordinator's blocked/ownership_ambiguous projection. Running resume correctly imports/blocks and returns ownership_ambiguous; typed retain applies only once blocked. Positive phase/dead-original-runner/same-attempt synchronization repair in isolated test only; independent review underway. No runtime/source edit from this observation." + "diagnosis": "Test kills attempt_runner, not live detached coordinator. Fixed.1s sleep does not establish coordinator's blocked/ownership_ambiguous projection. Running resume correctly imports/blocks and returns ownership_ambiguous; typed retain applies only once blocked. Positive phase/dead-original-runner/same-attempt synchronization repair in isolated test only; independent review underway. No runtime/source edit from this observation.", + "python314": { + "version": "3.14.7", + "job_id": 111099154216, + "tests": 604, + "seconds": 697.362, + "failures": 0, + "errors": 0, + "skips": 2, + "unraisable": [], + "status": "passed_on_prior_f40_not_next_candidate_acceptance" + } + }, + "delivery_phase_correction": { + "source_checkpoint": "893490a2dbd8fa4f48624d1ae0fe33ca106f63fe", + "test_blob": "7723f6ce3b23f715a4a20a38fe5bb59c1ce23bee", + "independent_review": "clean; committed source exactly reviewed, no product edits", + "agent_combined_modules": { + "tests": 52, + "python312_seconds": 82.966, + "python314_seconds": 86.973, + "failures": 0, + "errors": 0, + "skips": 0 + }, + "root_delivery": { + "tests": 19, + "seconds": 48.911, + "failures": 0, + "errors": 0, + "skips": 0 + }, + "unchanged_service_root_gate": "33/29.820s passed on exact same5456e21 test blob", + "core_unchanged": true, + "evidence": "evidence/release-repair-ownership-synchronization.json", + "root_final_bash": { + "assertions": 259, + "files": 11, + "focused_legacy_assertions": 50, + "failures": 0, + "exec_session": 89228, + "timings_seconds": { + "fast": 0.299, + "ignored_TERM": 0.228, + "pre_open_cancel": 0.225, + "TERM_resistant": 3.296, + "slow_ps": 1.323 + } + }, + "reference_json_staged_unstaged_diff": "passed", + "next_public_ci": "pending" + }, + "diagnostic_cleanup_overlap_triage": { + "source": "independent read-only code/output triage; no new probes", + "observation": "Rejected diagnostic RELEASE plus immediate cancellation produced ConflictError recovery cancellation is not active", + "classification": "fail-closed stale-phase completion fence; not evidence of duplicate writers or false cancellation success", + "possible_interleaving": "Concurrent importer can reset cancelling attempt to blocked while cancel_orphan completion requires cancelling/recovery_cleanup; no exact ordering claimed", + "missing_evidence": "No error-time DB/events snapshot; final rows or a lost-intent liveness defect cannot be established", + "disposition": "retain diagnostic concurrency-conflict / possible liveness follow-up, not confirmed unsafe product finding or passing gate; corrected helper awaits import DONE before fixture cleanup" } } diff --git a/docs/plans/engineering-team/evidence/release-repair-ownership-synchronization.json b/docs/plans/engineering-team/evidence/release-repair-ownership-synchronization.json new file mode 100644 index 0000000..f9d2746 --- /dev/null +++ b/docs/plans/engineering-team/evidence/release-repair-ownership-synchronization.json @@ -0,0 +1,88 @@ +{ + "schema_version": 1, + "recorded_on": "2026-10-02", + "scope": "Test-only positive delivery runner-death and blocked-ownership phase synchronization; no core edits or whole-suite rerun.", + "baseline_revision": "f40c618caefd100637c4686660673e159899ecb7", + "branch": "codex/release-repair-ownership-sync", + "preserved_branch": "codex/release-runner-exit-sync", + "test_blob": "7723f6ce3b23f715a4a20a38fe5bb59c1ce23bee", + "test_diff_sha256": "a3c19e9398adab6ac27526cf6d4e27a0d63fbf371cd576619c230259bf28eef5", + "public_CI_failure": { + "reported_by": "root", + "run_id": "37086958301", + "head_sha": "f40c618caefd100637c4686660673e159899ecb7", + "python": "3.11.9", + "tests_before_failfast_stop": 137, + "seconds": 191.313, + "failures": 1, + "errors": 0, + "skips": 0, + "unraisable": 0, + "test": "test_delivery_workflow.DeliveryWorkspaceTest.test_killed_repair_supervisor_never_launches_a_duplicate_writer", + "assertion": "ownership_ambiguous != retain_ownership at original line1001", + "other_jobs_at_assignment": "Bash and optional MCP passed; Python3.14 remained active. This agent did not cancel or mutate any CI job.", + "later_python314_outcome_reported_by_root": {"version": "3.14.7", "job_id": "111099154216", "tests": 604, "seconds": 697.362, "failures": 0, "errors": 0, "skips": 2, "unraisable": 0}, + "workflow_outcome": "failed because Python3.11.9 failed; Python3.14.7 success is not exact-next-head workflow acceptance" + }, + "cause": "Despite the historical test name, attempt.pid is the durable attempt_runner, NOT the detached coordinator or writer. After runner SIGKILL the surviving coordinator asynchronously reaps and imports, publishing blocked/recovery_required plus the original attempt's ownership_ambiguous status while the child may remain live. The test assumed .1s implied that phase was persisted. Service.resume called while still running first performs import_durable and returns ownership_ambiguous/launchedfalse; only the already-blocked branch validates and returns the typed retain_ownership disposition. The safety response is correct; the test's ordering assumption is not.", + "controlled_diagnostic": { + "helper_sha256": "498cebc1019a62551ff52f84318a88ca4fbcfbee14154c0e326cfe5d78431262", + "scope": "Private Popen scheduling wrapper only for frozen devsquad.detached coordinator argv. Exact frozen PYTHONPATH/package digest and worker commands remain unchanged. The spawned coordinator's in-memory import method pauses only on the original third implementer after wait_durable has reaped the killed runner, then calls its unchanged original importer.", + "synchronization": "Atomically publish exact ready JSON via temporary+replace. Parent establishes ready identity and actual original runner dead. Old test proceeds to one typed request while run is definitely running. Repaired pure status waiter first observes actual running, releases the private gate, then observes genuine coordinator blocked publication before its one typed request. Cleanup releases the gate, waits import DONE, and cancels the fixture before temporary removal. No numeric coordinator signals or classifier/state forcing.", + "original_source": "--original loads only the exact original method AST from git show f40c618caefd100637c4686660673e159899ecb7:test/core/test_delivery_workflow.py without modifying checkout files.", + "original_red": { + "python312": {"version": "3.12.14", "tests": 1, "seconds": 3.868, "failures": 1, "errors": 0}, + "python314": {"version": "3.14.6", "tests": 1, "seconds": 4.334, "failures": 1, "errors": 0}, + "observation_both": {"before_state": "running", "before_phase": null, "before_attempt_status": "running", "runner": "dead", "typed_calls": 1, "disposition": "ownership_ambiguous", "launched": false}, + "assertion": "same ownership_ambiguous != retain_ownership at original1001" + }, + "corrected_green": { + "python312": {"version": "3.12.14", "tests": 1, "seconds": 3.665, "failures": 0, "errors": 0}, + "python314": {"version": "3.14.6", "tests": 1, "seconds": 3.814, "failures": 0, "errors": 0}, + "observation_both": {"before_state": "blocked", "before_phase": "recovery_required", "before_attempt_status": "ownership_ambiguous", "runner": "dead", "typed_calls": 1, "disposition": "retain_ownership", "launched": false} + } + }, + "rejected_diagnostics_retained": [ + { + "method": "Stop coordinator before killing runner, then wait runner dead.", + "outcome": "Invalid phase seam: stopped parent cannot reap its runner zombie; inspector never reached confirmed dead. Failed1 fixture each on3.12.14/8.295s and3.14.6/8.401s, with no typed call. Cleanup resumed the owned coordinator and cancelled. Not claimed as matching CI red." + }, + { + "method": "First reaped-coordinator gate without waiting import DONE before cleanup cancellation.", + "outcome": "Matching assertion reproduced, but3.12 teardown concurrently resumed the coordinator importer and cancelled; cleanup raised ConflictError recovery cancellation is not active (1 expected failure+1 cleanup error/3.461s).3.14 had1 expected failure/no errors/3.569s. This unsuitable harness is not accepted as the clean red gate; diagnostic cleanup now positively waits DONE. The observed overlap is retained for separate root triage, without a core edit or broader product conclusion." + }, + { + "method": "Ready JSON initially used direct write_text.", + "outcome": "Independent review identified potential file-visible-before-full-JSON race; corrected to atomic temporary+replace before final repeated red/green. No product/test source change from this correction." + } + ], + "repair": { + "changed_file": "test/core/test_delivery_workflow.py only", + "phase_wait": "Replace fixed.1s sleep with bounded5s pure status/current-attempt/strong-runner-identity polling. Require dead original runner, blocked/recovery_required, same id/token/pid/pgid/start identity and ownership_ambiguous before the sole typed retain request.", + "bounds": "Original3s repair fixture and10s child-start wait are unchanged. No retry-until-success, alternative success disposition, provider delay widening or product timeout change.", + "preserved_and_strengthened": "No exit receipt; launchedfalse and exact retain_ownership; exactly implementer/reviewer/implementer attempts, original final id/token and3 worker invocations; run remains blocked until cancel; cancellation must be exactly cancelled; source checkout/index/HEAD/remotes remain unchanged. Comment distinguishes runner from coordinator without renaming the historical CI case." + }, + "affected_module_gates": { + "environment": "PYTHONDONTWRITEBYTECODE=1 PYTHONWARNINGS=error::ResourceWarning PYTHONPATH=plugin/core/src:test/core", + "command": "INTERPRETER -B -m unittest test_delivery_workflow test_service -q", + "modules": {"delivery_workflow": 19, "service": 33}, + "python312": {"version": "3.12.14", "tests": 52, "seconds": 82.966, "failures": 0, "errors": 0, "skips": 0}, + "python314": {"version": "3.14.6", "tests": 52, "seconds": 86.973, "failures": 0, "errors": 0, "skips": 0} + }, + "independent_review": { + "reviewer": "r6_readiness", + "source_review": "clean; pure phase poll and all same-attempt/no-new-writer/cancel/source assertions preserved or strengthened", + "real_cases": ["test_killed_repair_supervisor_never_launches_a_duplicate_writer", "test_live_implementer_cannot_be_resumed_into_a_second_writer", "test_cancelled_repair_retains_prior_candidate_attempts_and_disposition"], + "python312": {"version": "3.12.14", "tests": 3, "seconds": 10.469, "failures": 0, "errors": 0}, + "python314": {"version": "3.14.6", "tests": 3, "seconds": 11.505, "failures": 0, "errors": 0}, + "final_helper_review_and_gate": "Current corrected atomic-ready/DONE-ordered helper red/green independently confirmed; frozen product/test diff remains clean. No wider product conclusion inferred from rejected diagnostic cleanup overlap." + }, + "core_tree_unchanged": "62eea7fa31153ef732fd5e9dfd97951d95b367d8", + "other_gates": { + "bash": {"command": "PYTHONDONTWRITEBYTECODE=1 /bin/bash test/run.sh", "test_files": 11, "assertions": 259, "failures": 0, "legacy_assertions": 50}, + "watchdog_seconds": {"fast": 0.292, "ignored_TERM": 0.216, "early_cancel": 0.223, "tree": 2, "resistant": 3.257, "slow_probe": 1.306}, + "generated_reference": "current", + "JSON_and_diff": "pass" + }, + "limitations": "Local proof uses3.12.14 and3.14.6, not public3.11.9/3.14.7. No full604 suite, provider/native request, installation, publication, global settings or root edits. Root owns integration, separate diagnostic-overlap triage and the next exact-head public CI gate." +} diff --git a/test/core/test_delivery_workflow.py b/test/core/test_delivery_workflow.py index d5e67d6..7723f6c 100644 --- a/test/core/test_delivery_workflow.py +++ b/test/core/test_delivery_workflow.py @@ -22,6 +22,7 @@ ) from devsquad.service import Service from devsquad.store import ConflictError, Store, request_hash +from devsquad.supervisor import inspect_process from devsquad.workspaces import ( freeze_delivery_candidate, prepare_delivery_workspace, @@ -991,8 +992,27 @@ def test_killed_repair_supervisor_never_launches_a_duplicate_writer(self): time.sleep(0.02) if attempt is None: self.fail("repair writer did not publish its child identity") + # The attempt pid is the runner, not the still-live detached coordinator. os.kill(attempt["pid"], signal.SIGKILL) - time.sleep(0.1) + deadline = time.monotonic() + 5 + while time.monotonic() < deadline: + status = service.status(started["run_id"]) + store = Store(service.database, service.artifacts) + try: + current = store.attempt(started["run_id"]) + for field in ("id", "attempt_token", "pid", "pgid", "process_start_id"): + self.assertEqual(current[field], attempt[field]) + finally: + store.close() + if (inspect_process(attempt["pid"], attempt["pgid"], attempt["process_start_id"]) == "dead" + and status["state"] == "blocked" + and status["phase"] == "recovery_required" + and current["status"] == "ownership_ambiguous"): + break + time.sleep(0.02) + else: + self.fail("killed repair runner did not reach blocked ownership recovery") + self.assertFalse(Path(attempt["exit_record"]).exists()) recovered = Service(self.runtime).resume( started["run_id"], {"attempt_id": attempt["id"], "disposition": "retain_ownership"}, @@ -1006,6 +1026,10 @@ def test_killed_repair_supervisor_never_launches_a_duplicate_writer(self): [item["role"] for item in attempts], ["implementer", "reviewer", "implementer"], ) + self.assertEqual(attempts[-1]["id"], attempt["id"]) + self.assertEqual(attempts[-1]["attempt_token"], attempt["attempt_token"]) + self.assertEqual(store.worker_invocations(started["run_id"]), 3) + self.assertEqual(store.run(started["run_id"])["state"], "blocked") finally: store.close() cancelled = service.cancel(started["run_id"])